Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 025ed41d78557f162d403cbefd7209190c3291b4
parent 4782bea368f426a1f31f0133910023c725a6d0da
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 22 Jul 2026 11:00:57 -0400

Merge branch 'main' into worktree-feat+byo-ai-corpus

# Conflicts:
#	export/CHANGELOG.md

Diffstat:
Mcommon/components/SearchSessionContext.tsx | 18++++++++++++++++--
Mcommon/components/WorkspaceSearchBar.tsx | 163+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------
Mcommon/components/ui/checkbox.tsx | 7++++---
Mcommon/controller/autoRunner.ts | 3++-
Mcommon/controller/channelSnapshot.ts | 27+++++++++++++++++++++++++++
Mcommon/controller/checkAvailability.ts | 15+++++++++++++++
Mcommon/controller/persistKept.ts | 3++-
Acommon/jobs/jobSpec.test.ts | 50++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobSpec.ts | 4+++-
Mcommon/jobs/syncSchedulerState.ts | 19+++++++++++++++++--
Mcommon/lib/availability.ts | 12++++++++++++
Mcommon/lib/channelConfig.ts | 15+++++++++++++++
Mcommon/lib/channelGroups.ts | 11++++++++++-
Acommon/lib/cookiePolicy.test.ts | 109+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/cookiePolicy.ts | 86+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/downloadOutcome.ts | 6+++++-
Acommon/lib/momentUrl.ts | 180+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/settings.ts | 27+++++++++++++++++++++++++--
Mcommon/lib/transcriptToMarkdown.test.ts | 30++++++++++++++++++++++++++++++
Mcommon/lib/transcriptToMarkdown.ts | 29++++++++++++++++++++++++++++-
Mcommon/ytdlp/downloadOneManaged.ts | 134++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------
Mcommon/ytdlp/runYtdlp.ts | 130+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Meditor/CHANGELOG.md | 4++++
Aeditor/app/api/widget/sync/route.ts | 72++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/RetryBucketControl.tsx | 11++++++++++-
Meditor/app/channels/[slug]/components/stages/DownloadStage.tsx | 56++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/lib/stageStatus.ts | 1+
Meditor/app/channels/[slug]/page.tsx | 28++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/pipelineActions.ts | 20+++++++++++++++++---
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 5+++--
Meditor/app/channels/actions.ts | 82++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Meditor/app/channels/components/ChannelForm.tsx | 69++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------
Meditor/app/channels/components/ChannelFormClient.tsx | 26+++++++++++++++++++++-----
Meditor/app/channels/components/DeleteChannelForm.tsx | 1+
Aeditor/app/channels/components/SiteMembershipsSection.tsx | 192+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/components/parseChannelForm.ts | 11+++++++++++
Aeditor/app/channels/lib/siteMemberships.ts | 152+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/new/page.tsx | 14+++++++++++++-
Meditor/app/jobs/jobReplayRegistry.ts | 1+
Meditor/app/settings/actions.ts | 9+++++++++
Meditor/app/settings/components/SettingsForm.tsx | 33+++++++++++++++++++++++++++++++--
Meditor/app/sites/components/SiteForm.tsx | 26+++++++++++++++++++++++++-
Meditor/app/widget/components/MonitorWidget.tsx | 146+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Meditor/app/widget/components/WidgetConfigForm.tsx | 29+++++++++++++++++++++++++++++
Meditor/app/widget/components/WidgetControls.tsx | 137+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Meditor/app/widget/lib/config.ts | 35+++++++++++++++++++++++++++++++++++
Aeditor/e2e/channel-site-membership.spec.ts | 186+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/channels.spec.ts | 6++++--
Aeditor/e2e/cookies-mode.spec.ts | 331+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 42++++++++++++++++++++++++++++++++++++++----
Aeditor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/config.json | 5+++++
Aeditor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/playlist | 2++
Meditor/e2e/site-scope.spec.ts | 6+++++-
Meditor/e2e/sites-crud.spec.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/widget.spec.ts | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 6++++++
Aexport/e2e/channel-group-chips.spec.ts | 256+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aexport/e2e/inline-channel-chips.spec.ts | 240+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/README.md | 119++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Amcp/src/momentUrl.test.ts | 219+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/search.test.ts | 472+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mmcp/src/search.ts | 458++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/server.ts | 746++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Amcp/src/shareLink.test.ts | 168+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Amcp/src/shareLink.ts | 316+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/source.ts | 205++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/sourceController.test.ts | 20+++++++++++++++++++-
67 files changed, 5891 insertions(+), 252 deletions(-)

diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx @@ -909,11 +909,25 @@ function useSearchSessionState() { if (stored) { setProfiles(stored.profiles); setActiveProfileName(stored.activeProfileName); - setCollapsedGroups(new Set(stored.collapsedGroups ?? [])); if (typeof stored.filtersCollapsed === "boolean") { setFiltersCollapsed(stored.filtersCollapsed); } } + if (stored?.collapsedGroups !== undefined) { + // User preference honored (a stored [] — all groups open — stays). + setCollapsedGroups(new Set(stored.collapsedGroups)); + } else { + // Inline groups render as loose chips with no collapse affordance, so + // they're excluded from both the default-collapse set and the >1 + // heuristic (a lone collapsible group among inline chips stays open). + const collapsible = channelGroupings.filter((g) => !g.group.inline); + if (collapsible.length > 1) { + // Collapse state never touched on a multi-group site: start compact. + // The chips still show selection state, so nothing important is + // hidden. + setCollapsedGroups(new Set(collapsible.map((g) => g.group.id))); + } + } const search = typeof window !== "undefined" ? window.location.search : ""; const params = new URLSearchParams(search); @@ -1039,7 +1053,7 @@ function useSearchSessionState() { setHydrated(true); // eslint-disable-next-line react-hooks/exhaustive-deps - }, [channelOptions.length === 0, hydrated]); + }, [channelOptions.length === 0, hydrated, channelGroupings]); // Reflect the chart view + shape in the URL live (post-hydration only), so the // address bar always carries the current chart — mirrors how qt= is written. diff --git a/common/components/WorkspaceSearchBar.tsx b/common/components/WorkspaceSearchBar.tsx @@ -6,9 +6,12 @@ // (and keeps its committed search) across `/` ⇄ `/ask`. All of its state comes // from the shared SearchSession. +import { Fragment } from "react"; +import { ChevronRightIcon } from "lucide-react"; import { Button } from "./ui/button"; import { Checkbox } from "./ui/checkbox"; import QueryBuilder from "./QueryBuilder"; +import { cn } from "../lib/utils"; import { clearAdvanced } from "./exportAdvancedStorage"; import { ymdToInput, inputToYmd } from "../lib/ymd"; import { @@ -233,8 +236,38 @@ export default function WorkspaceSearchBar() { Reset everything </Button> </div> - <div className="flex flex-col gap-1.5"> + <div className="flex flex-wrap items-start gap-1.5"> {channelGroupings.map(({ group, channels }) => { + if (group.inline) { + // Inline group: no collapsible box — each member channel + // renders as its own loose chip flowing among the group + // chips, styled like a collapsed group chip. + return ( + <Fragment key={group.id}> + {channels.map((key) => { + const label = channelLabelByKey.get(key) ?? key; + return ( + <label + key={key} + title={label} + data-testid="inline-channel-chip" + className="flex items-center gap-2 rounded border border-border bg-card/60 px-2.5 py-1.5 text-sm cursor-pointer select-none min-w-0 max-w-full w-full sm:w-auto sm:min-w-48" + > + <span className="flex items-center shrink-0"> + <Checkbox + checked={!draftExcludedChannels.has(key)} + onCheckedChange={(value) => + setSelected([key], value === true) + } + /> + </span> + <span className="truncate">{label}</span> + </label> + ); + })} + </Fragment> + ); + } const isOpen = !collapsedGroups.has(group.id); const selectedCount = channels.reduce( (n, name) => @@ -245,17 +278,57 @@ export default function WorkspaceSearchBar() { <details key={group.id} open={isOpen} - onToggle={(e) => - toggleGroupCollapsed( - group.id, - (e.target as HTMLDetailsElement).open, - ) - } - className="rounded border border-border bg-card/60" + className={cn( + "rounded border border-border bg-card/60 min-w-0 max-w-full w-full", + !isOpen && "sm:w-auto sm:min-w-48", + )} > - <summary className="cursor-pointer select-none flex flex-wrap items-center gap-x-3 gap-y-1 px-3 py-1.5 text-sm"> + <summary + onClick={(e) => { + // Drive the open state from React rather than the + // browser's default toggle action — same rationale + // as the outer Filters summary above. + e.preventDefault(); + toggleGroupCollapsed(group.id, !isOpen); + }} + className="cursor-pointer select-none flex items-center gap-2 px-2.5 py-1.5 text-sm min-w-0" + title={!isOpen ? group.description : undefined} + > + <ChevronRightIcon + aria-hidden="true" + className={cn( + "size-3.5 shrink-0 text-muted-foreground transition-transform", + isOpen && "rotate-90", + )} + /> + <span + onClick={(e) => { + // Isolate checkbox clicks from the summary: no + // expand/collapse, just the group bulk-toggle. + e.preventDefault(); + e.stopPropagation(); + }} + className="flex items-center shrink-0" + > + <Checkbox + checked={ + selectedCount === 0 + ? false + : selectedCount === channels.length + ? true + : "indeterminate" + } + onCheckedChange={(value) => + setSelected(channels, value === true) + } + aria-label={`Select all in ${group.name || "group"}`} + /> + </span> {group.name && ( - <span className="font-medium text-foreground inline-flex items-center gap-1.5"> + <span + className="font-medium text-foreground flex items-center gap-1.5 min-w-0" + title={group.name} + > {group.accent && ( <span aria-hidden="true" @@ -263,40 +336,48 @@ export default function WorkspaceSearchBar() { style={{ background: group.accent }} /> )} - {group.name} + <span className="truncate">{group.name}</span> </span> )} - <span className="text-xs text-muted-foreground"> - ({selectedCount}/{channels.length} selected) - </span> - <span className="ml-auto flex items-center gap-2"> - <Button - type="button" - variant="link" - size="sm" - onClick={(e) => { - e.preventDefault(); - e.stopPropagation(); - setSelected(channels, true); - }} - className="text-xs text-muted-foreground" - > - All - </Button> - <Button - type="button" - variant="link" - size="sm" - onClick={(e) => { - e.preventDefault(); - e.stopPropagation(); - setSelected(channels, false); - }} - className="text-xs text-muted-foreground" - > - None - </Button> + <span + className={cn( + "text-xs text-muted-foreground shrink-0", + !isOpen && "ml-auto", + )} + > + {selectedCount}/{channels.length} + <span className="sr-only"> selected</span> </span> + {isOpen && ( + <span className="ml-auto flex items-center gap-2"> + <Button + type="button" + variant="link" + size="sm" + onClick={(e) => { + e.preventDefault(); + e.stopPropagation(); + setSelected(channels, true); + }} + className="text-xs text-muted-foreground" + > + All + </Button> + <Button + type="button" + variant="link" + size="sm" + onClick={(e) => { + e.preventDefault(); + e.stopPropagation(); + setSelected(channels, false); + }} + className="text-xs text-muted-foreground" + > + None + </Button> + </span> + )} </summary> {group.description && ( <div className="px-3 pt-0 pb-1 text-xs text-muted-foreground"> diff --git a/common/components/ui/checkbox.tsx b/common/components/ui/checkbox.tsx @@ -1,7 +1,7 @@ "use client" import * as React from "react" -import { CheckIcon } from "lucide-react" +import { CheckIcon, MinusIcon } from "lucide-react" import { Checkbox as CheckboxPrimitive } from "radix-ui" import { cn } from "../../lib/utils" @@ -14,7 +14,7 @@ function Checkbox({ <CheckboxPrimitive.Root data-slot="checkbox" className={cn( - "peer size-4 shrink-0 rounded-[4px] border border-input shadow-xs transition-shadow outline-none focus-visible:border-ring focus-visible:ring-[3px] focus-visible:ring-ring/50 disabled:cursor-not-allowed disabled:opacity-50 aria-invalid:border-destructive aria-invalid:ring-destructive/20 data-[state=checked]:border-primary data-[state=checked]:bg-primary data-[state=checked]:text-primary-foreground dark:bg-input/30 dark:aria-invalid:ring-destructive/40 dark:data-[state=checked]:bg-primary", + "peer group size-4 shrink-0 rounded-[4px] border border-input shadow-xs transition-shadow outline-none focus-visible:border-ring focus-visible:ring-[3px] focus-visible:ring-ring/50 disabled:cursor-not-allowed disabled:opacity-50 aria-invalid:border-destructive aria-invalid:ring-destructive/20 data-[state=checked]:border-primary data-[state=checked]:bg-primary data-[state=checked]:text-primary-foreground dark:bg-input/30 dark:aria-invalid:ring-destructive/40 dark:data-[state=checked]:bg-primary data-[state=indeterminate]:border-primary data-[state=indeterminate]:bg-primary data-[state=indeterminate]:text-primary-foreground dark:data-[state=indeterminate]:bg-primary", className )} {...props} @@ -23,7 +23,8 @@ function Checkbox({ data-slot="checkbox-indicator" className="grid place-content-center text-current transition-none" > - <CheckIcon className="size-3.5" /> + <CheckIcon className="size-3.5 group-data-[state=indeterminate]:hidden" /> + <MinusIcon className="hidden size-3.5 group-data-[state=indeterminate]:block" /> </CheckboxPrimitive.Indicator> </CheckboxPrimitive.Root> ) diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -32,6 +32,7 @@ import { pruneExpired, } from "../jobs/platformBackoff"; import { type DownloadFailureClass } from "../lib/availability"; +import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { type DownloadOutcomeStatus } from "../lib/downloadOutcome"; import { downloadQueueKey } from "../lib/queueKeys"; import { readChannelConfig } from "./channels"; @@ -622,7 +623,7 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { // The child job's own abort signal: registry.cancel(childJobId) aborts // it (→ kills yt-dlp) when the runner is hard-cancelled. signal, - globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + cookiePolicy: resolveCookiePolicy(settings, config), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, downloadFormatPreset: resolveDownloadFormatPreset({ diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -11,10 +11,13 @@ import { type VideoFiles, } from "../lib/videoStatus"; import { + AUTH_RETRY_CLASSES, AVAILABILITY_VALUES, EXCLUDED_FROM_DOWNLOAD, type Availability, } from "../lib/availability"; +import { resolveCookiePolicy } from "../lib/cookiePolicy"; +import { getSettings } from "../lib/settings"; import { loadAvailability, resolveEffectiveAvailability, @@ -106,6 +109,14 @@ export type ChannelSnapshot = { // user can re-download with a different format (e.g. Original). Optional: // older snapshots lack it; readers must default to []. shortAudio: string[]; + // Undownloaded playlist videos whose effective availability says browser + // cookies could recover them (AUTH_RETRY_CLASSES: needs_auth, members_only, + // private). Populated in EVERY cookie mode — members_only/private are + // batch-excluded regardless, and a subscribed/owning account's cookies can + // still fetch them — and drives the "Download with cookies" bucket button + // (retry-bucket with forceCookies). Optional: older snapshots lack it; + // readers must default to []. + needsCookies: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; @@ -339,7 +350,11 @@ export async function generateChannelSnapshot( // "Missing metadata" or "Failed transcriptions" creates noise the user // cannot resolve. const excludedById = new Map<string, Availability>(); + const effectiveById = new Map<string, Availability>(); for (const v of perVideo) { + if (v.effectiveAvailability) { + effectiveById.set(v.id, v.effectiveAvailability); + } if ( v.effectiveAvailability && (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes( @@ -535,7 +550,13 @@ export async function generateChannelSnapshot( excludedFromDownload.deleted.sort(); excludedFromDownload.private.sort(); + // The channel's resolved cookie mode. In "defer" mode, needs_auth videos are + // ALSO dropped from undownloadedIds (which the auto-download runner + // consumes), so they aren't re-attempted every run — they wait in the + // needsCookies bucket for the manual cookie run instead. + const cookiePolicy = resolveCookiePolicy(getSettings(), config ?? undefined); const undownloadedIds: string[] = []; + const needsCookies: string[] = []; for (const url of urls) { // Post-reconcile a video's dir is its canonical id, so the URL's canonical // id is the dir name directly. @@ -543,7 +564,12 @@ export async function generateChannelSnapshot( if (!dirId) continue; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; + const effective = effectiveById.get(dirId); + if (effective && AUTH_RETRY_CLASSES.has(effective)) { + needsCookies.push(dirId); + } if (excludedById.has(dirId)) continue; + if (cookiePolicy.mode === "defer" && effective === "needs_auth") continue; undownloadedIds.push(dirId); } @@ -589,6 +615,7 @@ export async function generateChannelSnapshot( skippedByFilter: skippedByFilter.sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), + needsCookies: needsCookies.sort(), }, undownloadedIds, excludedFromDownload, diff --git a/common/controller/checkAvailability.ts b/common/controller/checkAvailability.ts @@ -14,6 +14,12 @@ import { recordAvailability, resolveEffectiveAvailability, } from "../lib/availability-server"; +import { + alwaysCookies, + cookieArgs, + resolveCookiePolicy, +} from "../lib/cookiePolicy"; +import { getSettings } from "../lib/settings"; import { readChannelConfig } from "./channels"; import { resolveShardItems } from "./shard"; import type { Paths } from "../lib/paths"; @@ -115,6 +121,14 @@ export async function runAvailabilityCheck({ const dataDir = path.join(channelDir, "data"); const config = await readChannelConfig(paths, channelSlug); const extraArgs = config?.ytdlpExtraArgs ?? []; + // Probe cookies in "always" mode ONLY. when-required/defer probes stay + // cookie-free deliberately, so auth gating keeps being OBSERVED as + // needs_auth — defer mode's exclusion + Needs-cookies bucket depend on that + // signal. (Always-mode probes may report a cookie-recoverable video as + // public; that's the trade-off of prophylactic cookies.) + const probeCookies = alwaysCookies( + resolveCookiePolicy(getSettings(), config ?? undefined), + ); const limit = pLimit(concurrency && concurrency > 0 ? Math.floor(concurrency) : 1); @@ -201,6 +215,7 @@ export async function runAvailabilityCheck({ "--dump-json", "--skip-download", "--no-warnings", + ...cookieArgs(probeCookies), ...extraArgs, "--", url, diff --git a/common/controller/persistKept.ts b/common/controller/persistKept.ts @@ -2,6 +2,7 @@ import path from "node:path"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import { getSettings } from "../lib/settings"; +import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { isSavedVideo } from "../lib/savedVideo-server"; import { computeKeptVideoIds } from "./keptVideos"; import { findVideoSourceUrl } from "./undownloadedVideos"; @@ -99,7 +100,7 @@ export async function persistKept({ videoUrl: url, onLog: downloadLog, signal: downloadSignal, - globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + cookiePolicy: resolveCookiePolicy(settings, channelConfig), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, diff --git a/common/jobs/jobSpec.test.ts b/common/jobs/jobSpec.test.ts @@ -0,0 +1,50 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { parseJobSpec } from "./jobSpec"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/jobSpec.test.ts + +test("parses a minimal spec and rejects malformed input", () => { + assert.deepEqual(parseJobSpec({ kind: "sync", slug: "chan" }), { + kind: "sync", + slug: "chan", + }); + assert.equal(parseJobSpec(null), null); + assert.equal(parseJobSpec("sync"), null); + assert.equal(parseJobSpec({ kind: "", slug: "chan" }), null); + assert.equal(parseJobSpec({ kind: "sync" }), null); +}); + +test("accepts every replay bucket, including needsCookies", () => { + for (const bucket of [ + "partialDownloads", + "noTranscript", + "downloadedNoTranscript", + "incompleteTranscript", + "shortAudio", + "needsCookies", + ]) { + const spec = parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket }); + assert.equal(spec?.bucket, bucket); + } + // Unknown buckets invalidate the whole spec. + assert.equal( + parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket: "nope" }), + null, + ); +}); + +test("carries flag params verbatim (e.g. forceCookies)", () => { + const spec = parseJobSpec({ + kind: "retry-bucket", + slug: "chan", + bucket: "needsCookies", + params: { queueKey: "youtube", forceCookies: true }, + }); + assert.deepEqual(spec?.params, { queueKey: "youtube", forceCookies: true }); + // Array params are rejected (params must be a plain object). + assert.equal( + parseJobSpec({ kind: "sync", slug: "chan", params: [] })?.params, + undefined, + ); +}); diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts @@ -20,7 +20,8 @@ export type ReplayBucket = | "noTranscript" | "downloadedNoTranscript" | "incompleteTranscript" - | "shortAudio"; + | "shortAudio" + | "needsCookies"; export type JobSpec = { kind: string; @@ -40,6 +41,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([ "downloadedNoTranscript", "incompleteTranscript", "shortAudio", + "needsCookies", ]); // Defensive parse for a spec read back from JSON (a sidecar or the bookmarks diff --git a/common/jobs/syncSchedulerState.ts b/common/jobs/syncSchedulerState.ts @@ -46,6 +46,10 @@ export type SchedulerState = { // per-channel, job). Drives the savedVideoBackup.intervalMinutes cadence. // null = never run. See editor/app/scheduler/runTick.ts. lastSavedVideoBackupAt: number | null; + // Epoch ms of the last manual full "Sync all" sweep (syncAllChannelsAction). + // A global freshness marker distinct from any single channel's lastSyncedAt; + // surfaced by the monitor widget's last-sync readout. null = never run. + lastSyncAllAt: number | null; }; export const SCHEDULER_RUN_LOG_LIMIT = 50; @@ -63,7 +67,12 @@ export function emptyChannelSyncState(): ChannelSyncState { } export function emptySchedulerState(): SchedulerState { - return { channels: {}, runs: [], lastSavedVideoBackupAt: null }; + return { + channels: {}, + runs: [], + lastSavedVideoBackupAt: null, + lastSyncAllAt: null, + }; } // Read the state file, tolerating a missing/corrupt file by returning an empty @@ -89,7 +98,12 @@ export async function readSchedulerState(paths: Paths): Promise<SchedulerState> const runs: SchedulerRun[] = Array.isArray(r.runs) ? r.runs.map(coerceRun).slice(0, SCHEDULER_RUN_LOG_LIMIT) : []; - return { channels, runs, lastSavedVideoBackupAt: num(r.lastSavedVideoBackupAt) }; + return { + channels, + runs, + lastSavedVideoBackupAt: num(r.lastSavedVideoBackupAt), + lastSyncAllAt: num(r.lastSyncAllAt), + }; } // Write the state atomically (tmp file + rename), creating the .scheduler dir on @@ -102,6 +116,7 @@ export async function writeSchedulerState( channels: state.channels, runs: state.runs.slice(0, SCHEDULER_RUN_LOG_LIMIT), lastSavedVideoBackupAt: state.lastSavedVideoBackupAt ?? null, + lastSyncAllAt: state.lastSyncAllAt ?? null, }; await mkdir(path.dirname(paths.schedulerStateFile), { recursive: true }); const tmp = `${paths.schedulerStateFile}.tmp-${process.pid}`; diff --git a/common/lib/availability.ts b/common/lib/availability.ts @@ -31,6 +31,18 @@ export const EXCLUDED_FROM_DOWNLOAD: ReadonlyArray<Availability> = [ "private", ]; +// Availability classes a cookie (auth) retry can potentially recover — the +// gate for the managed downloader's auth-retry attempts and the membership +// rule for the per-channel "Needs cookies" snapshot bucket. Note the overlap +// with EXCLUDED_FROM_DOWNLOAD: members_only/private are batch-excluded, but +// cookies from a subscribed/owning account can still fetch them via the +// bucket's manual cookie run. +export const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([ + "needs_auth", + "members_only", + "private", +]); + export type AvailabilityHistorySource = "check" | "backfill" | "download"; export type AvailabilityHistoryEntry = { diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -3,6 +3,7 @@ import { isDownloadFormatPreset, type DownloadFormatPreset, } from "../ytdlp/downloadFormat"; +import { isCookieMode, type CookieMode } from "./cookiePolicy"; export type ChannelHandling = "youtube" | "transcribe"; @@ -96,6 +97,14 @@ export type ChannelConfig = { // omitted, the global SiteSettings value is used. Set false to allow this // channel to download currently-live/upcoming videos. skipLiveDownloads?: boolean; + // Per-channel override of the global cookies-from-browser browser spec + // (SiteSettings.cookiesFromBrowser). Omitted/empty = inherit the global + // value. See common/lib/cookiePolicy.ts. + cookiesFromBrowser?: string; + // Per-channel override of the global cookie mode + // (SiteSettings.cookieMode). Omitted = inherit. See + // common/lib/cookiePolicy.ts for the mode semantics. + cookieMode?: CookieMode; // Opt-in audio-integrity checking for sources that intermittently serve // corrupt audio mid-download (e.g. Odysee "original" format). When // enabled, the managed downloader periodically validates the in-progress @@ -233,6 +242,12 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null { if (typeof r.skipLiveDownloads === "boolean") { config.skipLiveDownloads = r.skipLiveDownloads; } + if (typeof r.cookiesFromBrowser === "string" && r.cookiesFromBrowser.trim()) { + config.cookiesFromBrowser = r.cookiesFromBrowser.trim(); + } + if (isCookieMode(r.cookieMode)) { + config.cookieMode = r.cookieMode; + } if ( typeof r.sleepBetweenDownloadsSeconds === "number" && Number.isFinite(r.sleepBetweenDownloadsSeconds) && diff --git a/common/lib/channelGroups.ts b/common/lib/channelGroups.ts @@ -15,16 +15,22 @@ export type ChannelGroup = { // group header so origin is legible in the channel filter. Absent in // single-site mode. accent?: string; + // Render this group's channels as loose individual chips instead of a + // collapsible group chip; absent = false. + inline?: boolean; }; export const DEFAULT_GROUP_FALLBACK_ID = "default"; // Synthesized fallback used when settings has no groups configured at all, -// so the UI always has a non-empty grouping to render. +// so the UI always has a non-empty grouping to render. Inline by default: +// with no configured grouping the channels render as loose individual chips +// rather than one big collapsible "All channels" box. export const FALLBACK_GROUP: ChannelGroup = { id: DEFAULT_GROUP_FALLBACK_ID, name: "All channels", selectedByDefault: true, + inline: true, }; const ID_RE = /^[a-z0-9][a-z0-9-]*$/; @@ -51,6 +57,9 @@ export function parseChannelGroup(raw: unknown): ChannelGroup | null { if (typeof r.order === "number" && Number.isFinite(r.order)) { group.order = Math.floor(r.order); } + // Absent/false stays absent so serialized JSON stays clean and existing + // configured groups default off. + if (r.inline === true) group.inline = true; return group; } diff --git a/common/lib/cookiePolicy.test.ts b/common/lib/cookiePolicy.test.ts @@ -0,0 +1,109 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + DEFAULT_COOKIE_MODE, + alwaysCookies, + authRetryCookies, + cookieArgs, + isCookieMode, + resolveCookiePolicy, + type ResolvedCookiePolicy, +} from "./cookiePolicy"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/cookiePolicy.test.ts +// (or `node_modules/.bin/tsx --test common/lib/cookiePolicy.test.ts` from the repo root) + +test("isCookieMode accepts exactly the three modes", () => { + assert.equal(isCookieMode("always"), true); + assert.equal(isCookieMode("when-required"), true); + assert.equal(isCookieMode("defer"), true); + assert.equal(isCookieMode(""), false); + assert.equal(isCookieMode("never"), false); + assert.equal(isCookieMode(undefined), false); + assert.equal(isCookieMode(null), false); + assert.equal(isCookieMode(42), false); +}); + +test("value resolution: channel > global > undefined; empty/whitespace = inherit", () => { + assert.equal( + resolveCookiePolicy({ cookiesFromBrowser: "firefox" }).cookies, + "firefox", + ); + assert.equal( + resolveCookiePolicy( + { cookiesFromBrowser: "firefox" }, + { cookiesFromBrowser: "chrome:Default" }, + ).cookies, + "chrome:Default", + ); + // Empty / whitespace channel value inherits the global. + assert.equal( + resolveCookiePolicy( + { cookiesFromBrowser: "firefox" }, + { cookiesFromBrowser: "" }, + ).cookies, + "firefox", + ); + assert.equal( + resolveCookiePolicy( + { cookiesFromBrowser: "firefox" }, + { cookiesFromBrowser: " " }, + ).cookies, + "firefox", + ); + // Nothing configured anywhere. + assert.equal(resolveCookiePolicy({}).cookies, undefined); + assert.equal(resolveCookiePolicy({ cookiesFromBrowser: " " }).cookies, undefined); + // Values are trimmed. + assert.equal( + resolveCookiePolicy({ cookiesFromBrowser: " firefox " }).cookies, + "firefox", + ); +}); + +test("mode resolution: channel > global > default", () => { + assert.equal(resolveCookiePolicy({}).mode, DEFAULT_COOKIE_MODE); + assert.equal(resolveCookiePolicy({ cookieMode: "always" }).mode, "always"); + assert.equal( + resolveCookiePolicy({ cookieMode: "always" }, { cookieMode: "defer" }).mode, + "defer", + ); + // Absent channel mode inherits the global. + assert.equal( + resolveCookiePolicy({ cookieMode: "defer" }, {}).mode, + "defer", + ); + assert.equal(resolveCookiePolicy({ cookieMode: "defer" }, null).mode, "defer"); + // Junk mode (e.g. hand-edited file that bypassed parsing) falls to default. + assert.equal( + resolveCookiePolicy({ cookieMode: "sometimes" as never }).mode, + DEFAULT_COOKIE_MODE, + ); +}); + +test("accessors encode the mode semantics", () => { + const p = (mode: ResolvedCookiePolicy["mode"], cookies?: string) => + ({ cookies, mode }) as ResolvedCookiePolicy; + + // always: cookies on every invocation, and on retries. + assert.equal(alwaysCookies(p("always", "firefox")), "firefox"); + assert.equal(authRetryCookies(p("always", "firefox")), "firefox"); + // when-required: never prophylactically, yes on auth retries. + assert.equal(alwaysCookies(p("when-required", "firefox")), undefined); + assert.equal(authRetryCookies(p("when-required", "firefox")), "firefox"); + // defer: never. + assert.equal(alwaysCookies(p("defer", "firefox")), undefined); + assert.equal(authRetryCookies(p("defer", "firefox")), undefined); + + // No value configured: every accessor degrades to undefined in every mode. + for (const mode of ["always", "when-required", "defer"] as const) { + assert.equal(alwaysCookies(p(mode)), undefined); + assert.equal(authRetryCookies(p(mode)), undefined); + } +}); + +test("cookieArgs emits the flag pair or nothing", () => { + assert.deepEqual(cookieArgs("firefox"), ["--cookies-from-browser", "firefox"]); + assert.deepEqual(cookieArgs(undefined), []); + assert.deepEqual(cookieArgs(""), []); +}); diff --git a/common/lib/cookiePolicy.ts b/common/lib/cookiePolicy.ts @@ -0,0 +1,86 @@ +// Client-safe cookie-policy resolution. No node-only imports — the editor's +// settings/channel forms are "use client" files that pull the mode constants +// in. Mirrors availability.ts in that respect. +// +// One browser-cookie spec (yt-dlp `--cookies-from-browser`) and one MODE decide +// how every yt-dlp spawn uses cookies: +// +// "always" — pass cookies on every invocation (enumeration, metadata +// prefetch, availability probes, downloads). +// "when-required" — (default; the historical behavior) cookies only to retry +// an attempt that failed with an auth/age error. Covers the +// download auth-retry AND the metadata-prefetch retry. +// "defer" — never use cookies in normal runs. Auth-gated videos are +// recorded (needs_auth), excluded from subsequent batch +// runs, and collected into the per-channel "Needs cookies" +// bucket whose button re-runs them with cookies forced on. +// +// The VALUE resolves channel-over-global (non-empty channel spec wins; empty = +// inherit). The MODE resolves the same way (absent channel mode = inherit). +// With no value configured, "always"/"when-required" degrade to the no-cookie +// behavior; "defer" still defers (the bucket run then warns that no cookie +// value is configured). + +export type CookieMode = "always" | "when-required" | "defer"; + +export const COOKIE_MODE_VALUES: ReadonlyArray<CookieMode> = [ + "always", + "when-required", + "defer", +]; + +export const DEFAULT_COOKIE_MODE: CookieMode = "when-required"; + +export function isCookieMode(v: unknown): v is CookieMode { + return v === "always" || v === "when-required" || v === "defer"; +} + +export type ResolvedCookiePolicy = { + // The resolved browser spec, or undefined when neither the channel nor the + // global settings carry a non-empty one. + cookies: string | undefined; + mode: CookieMode; +}; + +// The subset of SiteSettings / ChannelConfig this module reads. Structural, so +// neither settings.ts nor channelConfig.ts needs to be imported here (both +// import CookieMode from this file). +export type CookiePolicyInputs = { + cookiesFromBrowser?: string; + cookieMode?: CookieMode; +}; + +export function resolveCookiePolicy( + settings: CookiePolicyInputs, + channelConfig?: CookiePolicyInputs | null, +): ResolvedCookiePolicy { + const channelValue = channelConfig?.cookiesFromBrowser?.trim() ?? ""; + const globalValue = settings.cookiesFromBrowser?.trim() ?? ""; + const cookies = channelValue || globalValue || undefined; + const mode = isCookieMode(channelConfig?.cookieMode) + ? channelConfig!.cookieMode! + : isCookieMode(settings.cookieMode) + ? settings.cookieMode! + : DEFAULT_COOKIE_MODE; + return { cookies, mode }; +} + +// Cookies for an ordinary (non-retry) invocation: only "always" mode passes +// them prophylactically. Undefined in every other mode — including "defer", +// whose whole point is that normal runs stay cookie-free. +export function alwaysCookies(p: ResolvedCookiePolicy): string | undefined { + return p.mode === "always" ? p.cookies : undefined; +} + +// Cookies for retrying an attempt that failed with an auth/age error: +// available in "always" and "when-required", never in "defer" (the failure is +// recorded instead, feeding the Needs-cookies bucket). +export function authRetryCookies(p: ResolvedCookiePolicy): string | undefined { + return p.mode === "defer" ? undefined : p.cookies; +} + +// The argv fragment for a resolved cookie spec. Empty when there is none, so +// spawn sites can unconditionally spread it. +export function cookieArgs(cookies: string | undefined): string[] { + return cookies ? ["--cookies-from-browser", cookies] : []; +} diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts @@ -48,7 +48,11 @@ export type DownloadAttemptKind = | "audio-checked-primary" // The metadata-only pass that runs before the real download so app-level // filters can decide, and so the download can reuse it via --load-info-json. - | "metadata-prefetch"; + | "metadata-prefetch" + // A cookie re-run of a metadata prefetch that failed with an auth/age error + // (cookie mode "always"/"when-required" with a cookie value configured). + // Recorded with n: 0 alongside the failed prefetch it retries. + | "metadata-prefetch-auth-retry"; export type AudioCheckProbeVerdict = "clean" | "partial" | "malformed"; diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts @@ -0,0 +1,180 @@ +// Build a timestamped deep link for a cited moment in a video. +// +// Two shapes, in preference order: +// +// 1. **Archilyzer viewer link** — `${siteOrigin}/?v=<slug>&t=<seconds>`. This +// is the exact scheme `common/components/urlState.ts` reads: `v` is the +// video slug (`<channelSlug>/<id>`, the value the modal compares against +// `detail.slug`) and `t` is whole seconds. Opening it lands in the archive's +// transcript modal at the cited moment. Used whenever we know the public +// origin of the viewer that owns the video (a deployed site, or a hub +// member's url). +// +// 2. **Platform link** — the video's own `webpageUrl` plus a per-platform time +// parameter, mirroring the seek patterns the in-app players use +// (`common/components/{Odysee,Rumble,Twitch}Player.tsx`). Used as a fallback +// when there is no viewer origin (e.g. an MCP pointed at a local build). +// +// Pure and dependency-light so it can be reused by the MCP server, the browser +// report-citation UI, and build tools alike. + +import { detectPlatform, type Platform } from "./platform"; + +export type MomentUrlInput = { + // Public origin of the archilyzer viewer that owns this video (a RemoteSource + // base, or a hub member's url). When present and non-empty, we build a viewer + // deep link. Null/undefined ⇒ fall back to a platform link. + siteOrigin?: string | null; + // The video's slug as the viewer's `?v=` param expects it (`channelSlug/id`). + slug?: string | null; + // The cited moment, in seconds. + seconds: number; + // Fallback building blocks when there is no viewer origin. + webpageUrl?: string | null; + platform?: Platform | null; +}; + +// Twitch watch/VOD URLs take an `XhYmZs` time token, not raw seconds. +function twitchTime(totalSeconds: number): string { + const s = Math.max(0, Math.floor(totalSeconds)); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + return `${h}h${m}m${s % 60}s`; +} + +// Best-effort deep link into a platform's own watch page at `seconds`. Mirrors +// the per-platform time params the in-app players build. Platforms whose watch +// page has no reliable start param (Rumble, Kick) get the bare `webpageUrl`. +// Returns null only when there is no `webpageUrl` to work from. +export function platformMomentUrl( + webpageUrl: string | null | undefined, + platform: Platform | null | undefined, + seconds: number, +): string | null { + if (!webpageUrl) return null; + const secs = Math.max(0, Math.floor(seconds || 0)); + if (secs <= 0) return webpageUrl; + const plat = platform ?? detectPlatform(webpageUrl); + let u: URL; + try { + u = new URL(webpageUrl); + } catch { + return webpageUrl; + } + switch (plat) { + case "youtube": + u.searchParams.set("t", `${secs}s`); + return u.toString(); + case "odysee": + u.searchParams.set("t", String(secs)); + return u.toString(); + case "twitch": + u.searchParams.set("t", twitchTime(secs)); + return u.toString(); + // Rumble / Kick / unknown: the watch page has no dependable start param — + // return the plain webpage URL rather than an invalid seek. + default: + return webpageUrl; + } +} + +// The archilyzer viewer deep link, or null when `siteOrigin`/`slug` are missing +// or the origin can't be parsed. +export function viewerMomentUrl( + siteOrigin: string | null | undefined, + slug: string | null | undefined, + seconds: number, +): string | null { + if (!siteOrigin || !slug) return null; + const secs = Math.max(0, Math.floor(seconds || 0)); + let u: URL; + try { + u = new URL(siteOrigin); + } catch { + return null; + } + u.pathname = "/"; + u.search = ""; + u.hash = ""; + u.searchParams.set("v", slug); + if (secs > 0) u.searchParams.set("t", String(secs)); + return u.toString(); +} + +// The preferred moment link: viewer deep link when we have an origin+slug, +// else the platform fallback. Null when neither can be built. +export function momentUrl(input: MomentUrlInput): string | null { + const viewer = viewerMomentUrl(input.siteOrigin, input.slug, input.seconds); + if (viewer) return viewer; + return platformMomentUrl(input.webpageUrl, input.platform, input.seconds); +} + +// ─── Base (appendable) forms ─── +// +// A "moment base" is a URL that ends in `t=`, so appending integer seconds +// yields a valid moment link (`<base>156` ≡ the momentUrl for 156s). Used by +// the MCP's compact `link_style:"base"` output: one base per video instead of +// a full URL per line, expanded back to full links in the final report. + +// The viewer deep-link base: same normalization as viewerMomentUrl, with `t` +// set last and empty so the caller can append seconds. Null when the origin or +// slug is missing/unparseable. +export function viewerMomentBaseUrl( + siteOrigin: string | null | undefined, + slug: string | null | undefined, +): string | null { + if (!siteOrigin || !slug) return null; + let u: URL; + try { + u = new URL(siteOrigin); + } catch { + return null; + } + u.pathname = "/"; + u.search = ""; + u.hash = ""; + u.searchParams.set("v", slug); + u.searchParams.set("t", ""); + return u.toString(); +} + +// The platform base. Unlike platformMomentUrl (which falls back to the bare +// webpage URL), a base MUST be appendable — so only platforms whose time param +// takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs` +// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown +// have no dependable start param at all. Null in every non-appendable case. +export function platformMomentBaseUrl( + webpageUrl: string | null | undefined, + platform: Platform | null | undefined, +): string | null { + if (!webpageUrl) return null; + const plat = platform ?? detectPlatform(webpageUrl); + let u: URL; + try { + u = new URL(webpageUrl); + } catch { + return null; + } + switch (plat) { + case "youtube": // accepts raw seconds in `t` (the `s` suffix is optional) + case "odysee": { + // delete-then-set so a pre-existing `t` is re-appended LAST — the base + // must end in `t=` for the append rule to hold. + u.searchParams.delete("t"); + u.searchParams.set("t", ""); + return u.toString(); + } + default: + return null; + } +} + +// The preferred moment base: viewer first, platform fallback — mirroring +// momentUrl's preference order. Null when neither can be built. +export function momentBaseUrl( + input: Omit<MomentUrlInput, "seconds">, +): string | null { + const viewer = viewerMomentBaseUrl(input.siteOrigin, input.slug); + if (viewer) return viewer; + return platformMomentBaseUrl(input.webpageUrl, input.platform); +} diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -24,6 +24,11 @@ import { defaultAutoQueue, sanitizeAutoQueue, } from "../jobs/autoQueuePolicy"; +import { + DEFAULT_COOKIE_MODE, + isCookieMode, + type CookieMode, +} from "./cookiePolicy"; export type { Worker } from "./workers"; export type { AutoQueueSettings } from "../jobs/autoQueuePolicy"; @@ -63,9 +68,18 @@ export type SiteSettings = { // app (see defaultWorkersFromApps). See common/lib/workers.ts. workers: Worker[]; // Browser spec (e.g. "firefox", "chrome:Default") passed to - // `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the - // managed per-video download flow. Empty string = disabled. + // `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by + // `cookieMode` below. Empty string = no cookies configured. Per-channel + // override available (ChannelConfig.cookiesFromBrowser). cookiesFromBrowser: string; + // How yt-dlp invocations use the configured cookies (see + // common/lib/cookiePolicy.ts): "always" passes them on every invocation, + // "when-required" (default; the historical behavior) only to retry an + // auth/age failure, "defer" never in normal runs — auth-gated videos are + // excluded from batches and collected into the per-channel "Needs cookies" + // bucket for a manual cookie run. Per-channel override available + // (ChannelConfig.cookieMode). + cookieMode: CookieMode; // Pause (seconds) inserted between per-video yt-dlp invocations in // managed batch downloads. yt-dlp's own `-t sleep` only paces requests // within one invocation, so without this the managed loop hammers the @@ -455,6 +469,7 @@ function defaults(): SiteSettings { transcriptionApps: {}, workers: [], cookiesFromBrowser: "", + cookieMode: DEFAULT_COOKIE_MODE, sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, downloadFormat: "auto", minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT, @@ -635,6 +650,11 @@ export function getSettings(): SiteSettings { if (typeof merged.cookiesFromBrowser !== "string") { merged.cookiesFromBrowser = ""; } + // A settings.json predating cookieMode (or carrying junk) gets the default, + // which preserves the historical retry-only behavior. + if (!isCookieMode(merged.cookieMode)) { + merged.cookieMode = DEFAULT_COOKIE_MODE; + } merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds( merged.sleepBetweenDownloadsSeconds, ); @@ -826,6 +846,9 @@ export async function writeSettings(next: SiteSettings): Promise<void> { typeof next.cookiesFromBrowser === "string" ? next.cookiesFromBrowser.trim() : "", + cookieMode: isCookieMode(next.cookieMode) + ? next.cookieMode + : DEFAULT_COOKIE_MODE, sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds( next.sleepBetweenDownloadsSeconds, ), diff --git a/common/lib/transcriptToMarkdown.test.ts b/common/lib/transcriptToMarkdown.test.ts @@ -48,3 +48,33 @@ test("maxCues truncates and notes it", () => { assert.doesNotMatch(md, /General Kenobi/); assert.match(md, /truncated: showing 1 of 2 cues/); }); + +test("stampForCue renders the compact form and floors fractional seconds", () => { + const md = transcriptToMarkdown(base, { + stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`, + }); + assert.match(md, /^\[0:00\|0\] Hello there$/m); + // 3661.25s floors to 3661 — the same integer momentUrl would use, so the + // compact stamp and the inline link cite the identical second. + assert.match(md, /^\[1:01:01\|3661\] General Kenobi$/m); +}); + +test("stampForCue takes precedence over linkForCue (no inline links)", () => { + const md = transcriptToMarkdown(base, { + stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`, + linkForCue: (seconds) => `https://example.test/abc123?t=${seconds}s`, + }); + assert.ok(!md.includes("](https://example.test/"), "no per-line links"); + assert.match(md, /^\[0:00\|0\] Hello there$/m); +}); + +test("extraMeta lines render as `- <line>` in the metadata header", () => { + const md = transcriptToMarkdown(base, { + extraMeta: ["moment_base: https://example.test/abc123?t="], + }); + assert.match(md, /^- moment_base: https:\/\/example\.test\/abc123\?t=$/m); + assert.ok( + md.indexOf("moment_base:") < md.indexOf("## Description"), + "extra meta sits in the header, before the body", + ); +}); diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts @@ -31,6 +31,19 @@ export type TranscriptMarkdownOptions = { // Cap the number of cue lines emitted (for fitting a context window). When // truncated, a marker line is appended. Default: no cap. maxCues?: number; + // Optional builder turning a cue's start seconds into a deep link. When it + // returns a URL, the timestamp is rendered as a Markdown link + // (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when + // `timestamps` is false or when it returns null. See `momentUrl`. + linkForCue?: (seconds: number) => string | null; + // Optional formatter for the bracketed label's CONTENTS (e.g. "2:36|156" → + // rendered as `[2:36|156] text`). Receives the pre-formatted clock and the + // cue's raw start seconds. Takes precedence over `linkForCue`. Ignored when + // `timestamps` is false. + stampForCue?: (clock: string, seconds: number) => string; + // Extra metadata lines rendered as `- <line>` after the tags line (e.g. + // "moment_base: <url>"). + extraMeta?: string[]; }; // [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so @@ -53,6 +66,9 @@ export function transcriptToMarkdown( includeDescription = true, includeTags = false, maxCues, + linkForCue, + stampForCue, + extraMeta, } = options; const lines: string[] = []; @@ -70,6 +86,7 @@ export function transcriptToMarkdown( if (includeTags && input.tags && input.tags.length > 0) { meta.push(`- Tags: ${input.tags.join(", ")}`); } + for (const line of extraMeta ?? []) meta.push(`- ${line}`); lines.push(...meta); if (includeDescription && input.description && input.description.trim()) { @@ -97,7 +114,17 @@ export function transcriptToMarkdown( const cue = cues[i]; const text = cue.text.trim(); if (!text) continue; - lines.push(timestamps ? `[${stamp(cue.start)}] ${text}` : text); + if (!timestamps) { + lines.push(text); + continue; + } + if (stampForCue) { + lines.push(`[${stampForCue(stamp(cue.start), cue.start)}] ${text}`); + continue; + } + const url = linkForCue ? linkForCue(cue.start) : null; + const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start); + lines.push(`[${label}] ${text}`); } if (limit < cues.length) { lines.push(""); diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -3,11 +3,17 @@ import { appendFile, mkdir, readdir, readFile, rm } from "node:fs/promises"; import { createWriteStream, type WriteStream } from "node:fs"; import { execa } from "execa"; import { + AUTH_RETRY_CLASSES, classifyDownloadFailure, parseUnavailableFromStderr, - type Availability, } from "../lib/availability"; import { + DEFAULT_COOKIE_MODE, + alwaysCookies, + authRetryCookies, + type ResolvedCookiePolicy, +} from "../lib/cookiePolicy"; +import { type AudioFormat, type ChannelConfig, } from "../lib/channelConfig"; @@ -82,9 +88,12 @@ export type ManagedDownloadOpts = { videoUrl: string; onLog: (s: string) => void; signal: AbortSignal; - // Resolved global setting; only used when the primary attempt fails with an - // auth/age error AND the channel doesn't already have its own cookies set. - globalCookiesFromBrowser?: string; + // Resolved cookie policy (value + mode), already collapsed channel-over- + // global by the caller (resolveCookiePolicy). Governs which attempts pass + // --cookies-from-browser: "always" on every invocation, "when-required" + // (the default when omitted) only on auth-retry passes, "defer" never — + // failures are recorded for the Needs-cookies bucket instead. + cookiePolicy?: ResolvedCookiePolicy; // When false, suppress appending to the channel archive file. Mirrors // `ignoreArchive` from the playlist-level callers. appendArchive?: boolean; @@ -349,12 +358,6 @@ function attemptSucceeded(exitCode: number | null): boolean { return exitCode === 0 || exitCode === 101; } -const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([ - "needs_auth", - "members_only", - "private", -]); - async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> { const entries = await readdir(videoDir).catch(() => [] as string[]); return entries.some((e) => { @@ -467,6 +470,17 @@ async function runManagedDownload( const attempts: DownloadAttempt[] = []; let status: DownloadOutcomeStatus = "failed"; let fellBackToTranscribe = false; + // Cookie policy for every yt-dlp spawn in this download. An omitted policy + // behaves like the historical default: no prophylactic cookies, retry-only + // (and with no value configured the retries degrade to no-ops). + const cookiePolicy: ResolvedCookiePolicy = opts.cookiePolicy ?? { + cookies: undefined, + mode: DEFAULT_COOKIE_MODE, + }; + // Set when a failed metadata prefetch was recovered by its cookie retry: the + // real download then gets cookies immediately (skipping a doomed cookie-less + // attempt) and a success is reported as "ok-with-cookies". + let prefetchNeededCookies = false; // Set when the download-time duration guard trips: the measured shortfall, // recorded on the outcome so the UI can explain it without re-probing. let shortAudioInfo: NonNullable<DownloadOutcomeRecord["shortAudio"]> | undefined; @@ -498,7 +512,10 @@ async function runManagedDownload( opts.channelConfig.platform ?? detectPlatform(opts.videoUrl); if (canonicalId && !opts.signal.aborted) { const videoDir = path.join(channelDir, "data", canonicalId); - const prefetchArgs = [ + // In "always" mode the prefetch (like every other invocation) carries the + // configured cookies up front. + const prefetchCookies = alwaysCookies(cookiePolicy); + const buildPrefetchArgs = (cookies: string | undefined) => [ "--ignore-config", "--restrict-filenames", ...outputArgsForUrl(opts.videoUrl), @@ -506,11 +523,15 @@ async function runManagedDownload( "--skip-download", "--no-write-subs", "--no-write-auto-subs", - ...channelConfigArgs(opts.channelConfig), + ...channelConfigArgs(opts.channelConfig, cookies), "--", opts.videoUrl, ]; - const prefetchRes = await runOneYtdlp(opts, channelDir, prefetchArgs); + const prefetchRes = await runOneYtdlp( + opts, + channelDir, + buildPrefetchArgs(prefetchCookies), + ); lastFullTail = prefetchRes.stderrTail; const prefetchAvail = attemptSucceeded(prefetchRes.exitCode) ? undefined @@ -519,7 +540,7 @@ async function runManagedDownload( n: 0, kind: "metadata-prefetch", handling: opts.channelConfig.handling, - usedCookies: false, + usedCookies: Boolean(prefetchCookies), ytdlpExitCode: prefetchRes.exitCode, availabilityClass: prefetchAvail, error: attemptSucceeded(prefetchRes.exitCode) @@ -527,6 +548,49 @@ async function runManagedDownload( : trimError(prefetchRes.stderrTail), }); + // Prefetch auth retry: a prefetch that failed with an auth/age error is + // re-run once with cookies (when the mode allows it and the failed pass + // didn't already use them). A recovery here lets the real download start + // with cookies immediately instead of burning a doomed cookie-less + // attempt first. Defer mode lands here with no retry cookies, so the + // failure stands and feeds the Needs-cookies bucket. + const prefetchRetryCookies = authRetryCookies(cookiePolicy); + if ( + !attemptSucceeded(prefetchRes.exitCode) && + prefetchAvail !== undefined && + AUTH_RETRY_CLASSES.has(prefetchAvail) && + prefetchRetryCookies !== undefined && + !prefetchCookies && + !opts.signal.aborted + ) { + opts.onLog( + `Metadata prefetch auth-required (${prefetchAvail}); retrying with --cookies-from-browser ${prefetchRetryCookies}\n`, + ); + const retryRes = await runOneYtdlp( + opts, + channelDir, + buildPrefetchArgs(prefetchRetryCookies), + ); + lastFullTail = retryRes.stderrTail; + const retryAvail = attemptSucceeded(retryRes.exitCode) + ? undefined + : parseUnavailableFromStderr(retryRes.stderrTail); + attempts.push({ + n: 0, + kind: "metadata-prefetch-auth-retry", + handling: opts.channelConfig.handling, + usedCookies: true, + ytdlpExitCode: retryRes.exitCode, + availabilityClass: retryAvail, + error: attemptSucceeded(retryRes.exitCode) + ? undefined + : trimError(retryRes.stderrTail), + }); + if (attemptSucceeded(retryRes.exitCode)) { + prefetchNeededCookies = true; + } + } + const metaPath = path.join(videoDir, "metadata.info.json"); const metadata = await loadRawMetadata(metaPath); // Only wire --load-info-json into the real attempts when we actually have @@ -629,6 +693,13 @@ async function runManagedDownload( } } + // Cookies for the real download attempts: always-mode passes them on every + // invocation; otherwise a cookie-recovered prefetch means this video needs + // them, so don't burn a doomed cookie-less attempt first. + const primaryCookies = + alwaysCookies(cookiePolicy) ?? + (prefetchNeededCookies ? cookiePolicy.cookies : undefined); + let primaryRes: AttemptOutcome; let audioCheckStats: AudioCheckAttemptStats | undefined; let audioCheckCorruptSource = false; @@ -652,7 +723,7 @@ async function runManagedDownload( ), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, - ...channelConfigArgs(opts.channelConfig), + ...channelConfigArgs(opts.channelConfig, primaryCookies), "--", opts.videoUrl, ]; @@ -696,7 +767,7 @@ async function runManagedDownload( ...mediaArgs, "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, - ...channelConfigArgs(opts.channelConfig), + ...channelConfigArgs(opts.channelConfig, primaryCookies), ...sourceArgs(opts.videoUrl, infoJsonPath), ]; primaryRes = await runOneYtdlp(opts, channelDir, primaryArgs); @@ -710,7 +781,7 @@ async function runManagedDownload( n: 1, kind: audioCheckEnabled ? "audio-checked-primary" : "primary", handling: opts.channelConfig.handling, - usedCookies: false, + usedCookies: Boolean(primaryCookies), ytdlpExitCode: primaryRes.exitCode, availabilityClass: primaryAvail, error: attemptSucceeded(primaryRes.exitCode) @@ -725,7 +796,14 @@ async function runManagedDownload( !audioCheckCorruptFullSource && attemptSucceeded(primaryRes.exitCode); if (lastSucceeded) { - status = audioCheckEnabled ? "ok-audio-checked" : "ok"; + // "ok-with-cookies" keeps meaning "cookies were NEEDED" (the prefetch only + // succeeded with them) — always-mode prophylactic cookies on a clean + // success stay a plain "ok". + status = audioCheckEnabled + ? "ok-audio-checked" + : prefetchNeededCookies + ? "ok-with-cookies" + : "ok"; } else if (audioCheckCorruptSource) { status = "failed-corrupt-source"; } else if (audioCheckCorruptFullSource) { @@ -735,15 +813,20 @@ async function runManagedDownload( } // ---------- Attempt 2: auth retry ---------- + // Off in defer mode (authRetryCookies -> undefined), so the failure is + // recorded as-is and feeds the Needs-cookies bucket. Also skipped when the + // failed attempt already used cookies — retrying identically is futile. + const downloadRetryCookies = authRetryCookies(cookiePolicy); const shouldAuthRetry = !lastSucceeded && primaryAvail !== undefined && AUTH_RETRY_CLASSES.has(primaryAvail) && - Boolean(opts.globalCookiesFromBrowser); + downloadRetryCookies !== undefined && + !primaryCookies; if (shouldAuthRetry && !opts.signal.aborted) { opts.onLog( - `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${opts.globalCookiesFromBrowser}\n`, + `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${downloadRetryCookies}\n`, ); const retryArgs = [ "--ignore-config", @@ -752,7 +835,7 @@ async function runManagedDownload( ...mediaArgs, "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, - ...channelConfigArgs(opts.channelConfig, opts.globalCookiesFromBrowser), + ...channelConfigArgs(opts.channelConfig, downloadRetryCookies), ...sourceArgs(opts.videoUrl, infoJsonPath), ]; const retryRes = await runOneYtdlp(opts, channelDir, retryArgs); @@ -862,11 +945,12 @@ async function runManagedDownload( ...opts.channelConfig, handling: "transcribe", }; - // Use cookies if the auth retry succeeded with them; otherwise channel-level. + // Cookies for the fallback: always-mode passes them like every other + // invocation; otherwise only when this video demonstrably needed them + // (an attempt succeeded with cookies -> "ok-with-cookies"). const fallbackCookieOverride = - status === "ok-with-cookies" - ? opts.globalCookiesFromBrowser - : undefined; + alwaysCookies(cookiePolicy) ?? + (status === "ok-with-cookies" ? cookiePolicy.cookies : undefined); // Feed yt-dlp the metadata it already wrote during the primary // attempt instead of re-querying the extractor — saves a network // round-trip per video, which adds up across batch runs and helps diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -20,6 +20,12 @@ import { classifyDownloadFailure, type Availability, } from "../lib/availability"; +import { + alwaysCookies, + cookieArgs, + resolveCookiePolicy, + type ResolvedCookiePolicy, +} from "../lib/cookiePolicy"; import { resolveEffectiveAvailability } from "../lib/availability-server"; import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability"; import { resolveShardItems } from "../controller/shard"; @@ -96,6 +102,11 @@ export type RunYtdlpOpts = { // saved playlist by extracted ID). IDs not present in the playlist are // reported and skipped. bucketIds?: ReadonlyArray<string>; + // retry-bucket only (the "Needs cookies" bucket button): force cookie mode + // "always" for this run, so every invocation carries the configured cookie + // value and the defer-mode batch exclusion is bypassed. Warns (and behaves + // like a plain run) when no cookie value is configured. + forceCookies?: boolean; // retry-bucket only: override channelConfig.handling for this run without // mutating the channel config on disk. Useful for retrying old "youtube" // videos as "transcribe". @@ -170,12 +181,46 @@ function abortableSleep(ms: number, signal: AbortSignal): Promise<void> { }); } -function configArgs(config: ChannelConfig): string[] { +function configArgs(config: ChannelConfig, cookies?: string): string[] { const args: string[] = []; + args.push(...cookieArgs(cookies)); if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs); return args; } +// The run-level cookie policy: settings + channel overrides, with the +// retry-bucket forceCookies flag overriding the mode to "always" (that's the +// "Download with cookies" button). Logs a warning when cookies are forced but +// no value is configured anywhere — the run then proceeds cookie-less. +function resolveRunCookiePolicy( + opts: RunYtdlpOpts, + channelConfig: ChannelConfig, +): ResolvedCookiePolicy { + const policy = resolveCookiePolicy(getSettings(), channelConfig); + if (!opts.forceCookies) return policy; + if (!policy.cookies) { + opts.onLog( + "Force-cookies run requested, but no cookies-from-browser value is configured (global or channel). Proceeding without cookies.\n", + ); + } + return { ...policy, mode: "always" }; +} + +// Defer-mode batch exclusion: a video whose effective availability is +// needs_auth is skipped by batch runs when the resolved cookie mode is +// "defer" (it lands in the snapshot's Needs-cookies bucket instead of being +// re-attempted on every sync/download-missing). members_only/private are +// already covered by EXCLUDED_FROM_DOWNLOAD. forceCookies runs resolve to +// mode "always" and therefore bypass this. +export async function isDeferredAuthExcluded( + videoDir: string, + policy: ResolvedCookiePolicy, +): Promise<boolean> { + if (policy.mode !== "defer") return false; + const cls = await resolveEffectiveAvailability(videoDir); + return cls === "needs_auth"; +} + const OUTPUT_ARGS: string[] = [ "-o", "data/%(id)s/audio.%(ext)s", @@ -247,7 +292,15 @@ async function enumeratePlaylistUrls( if (range) { args.push("--lazy-playlist", "-I", `${range.start}:${range.end}`); } - args.push(...configArgs(opts.channelConfig), opts.channelConfig.url!); + // Cookie mode "always" covers enumeration too (a members-only or otherwise + // gated channel may not even list without cookies). + args.push( + ...configArgs( + opts.channelConfig, + alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)), + ), + opts.channelConfig.url!, + ); opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`); const child = execa(opts.paths.ytdlpBin, args, { @@ -436,8 +489,14 @@ async function downloadPlaylistManaged( // Exclude videos whose effective availability marks them as permanently // unavailable (members_only, deleted, private). Applies to both prefilter // modes: failed download attempts don't write to the archive, so the - // archive-prefilter path would also keep retrying them. + // archive-prefilter path would also keep retrying them. Under cookie mode + // "defer", needs_auth videos are excluded too — they wait in the snapshot's + // Needs-cookies bucket for a manual cookie run instead of being re-attempted + // (and re-failing) on every batch. A forceCookies run resolves to mode + // "always" and so bypasses the defer exclusion. + const runCookiePolicy = resolveRunCookiePolicy(opts, effectiveChannelConfig); const excludedCounts = { members_only: 0, deleted: 0, private: 0 }; + let deferredAuthCount = 0; const filteredTofetch: string[] = []; for (const url of tofetch) { const dirId = extractVideoId(url); @@ -452,6 +511,10 @@ async function downloadPlaylistManaged( excludedCounts[cls as keyof typeof excludedCounts]++; continue; } + if (runCookiePolicy.mode === "defer" && cls === "needs_auth") { + deferredAuthCount++; + continue; + } } filteredTofetch.push(url); } @@ -464,6 +527,11 @@ async function downloadPlaylistManaged( `Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`, ); } + if (deferredAuthCount > 0) { + opts.onLog( + `Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`, + ); + } tofetch.length = 0; tofetch.push(...filteredTofetch); @@ -539,7 +607,12 @@ async function downloadPlaylistManaged( } const { okCount, failedCount, skippedCount, processedCount, firstFailure } = - await runManagedDownloads(opts, items, effectiveChannelConfig); + await runManagedDownloads( + opts, + items, + effectiveChannelConfig, + runCookiePolicy, + ); if (firstFailure && !opts.signal.aborted) { throw firstFailure; } @@ -575,9 +648,15 @@ async function runManagedDownloads( opts: RunYtdlpOpts, urls: ReadonlyArray<string>, effectiveChannelConfig: ChannelConfig, + // Cookie policy for every managed download in this run. Callers that + // already resolved it (for their own prefilter) pass it through so the + // forceCookies-without-value warning isn't logged twice. + cookiePolicyOverride?: ResolvedCookiePolicy, ): Promise<ManagedRunResult> { const settings = getSettings(); - const globalCookies = settings.cookiesFromBrowser; + const cookiePolicy = + cookiePolicyOverride ?? + resolveRunCookiePolicy(opts, effectiveChannelConfig); const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback; const globalSkipLiveDownloads = settings.skipLiveDownloads; // Resolve the download format once per run (override > channel > global), @@ -644,7 +723,7 @@ async function runManagedDownloads( videoUrl: url, onLog: task ? task.onLog : opts.onLog, signal: opts.signal, - globalCookiesFromBrowser: globalCookies || undefined, + cookiePolicy, appendArchive: !opts.ignoreArchive, inlineTranscribeOnFallback, globalSkipLiveDownloads, @@ -782,7 +861,10 @@ async function downloadSubsForUrl( "--skip-download", "--no-write-info-json", "--no-overwrites", - ...configArgs(opts.channelConfig), + ...configArgs( + opts.channelConfig, + alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)), + ), "--", url, ]; @@ -951,7 +1033,10 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { "--audio-format", fmt, ...(opts.channelConfig.keepSourceVideo ? ["-k"] : []), - ...configArgs(opts.channelConfig), + ...configArgs( + opts.channelConfig, + alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)), + ), ...(opts.extraYtdlpArgs ?? []), "--", opts.singleVideoUrl, @@ -1013,6 +1098,9 @@ async function sync(opts: RunYtdlpOpts): Promise<void> { let totalFailed = 0; let totalSkipped = 0; let firstFailure: Error | null = null; + // Resolved once for the whole sync: drives both the defer-mode needs_auth + // exclusion below and the managed downloads themselves. + const runCookiePolicy = resolveRunCookiePolicy(opts, opts.channelConfig); for ( let page = 0; @@ -1026,20 +1114,44 @@ async function sync(opts: RunYtdlpOpts): Promise<void> { const newUrls: string[] = []; let archivedHits = 0; + let deferredAuthCount = 0; for (const url of pageUrls) { const archiveId = await archiveIdForUrl(url, dataDir); if (archiveId && archive.ids.has(archiveId)) { archivedHits++; continue; } + // Cookie mode "defer": don't re-attempt known auth-gated videos on + // every sync — they wait in the Needs-cookies bucket instead. + const dirId = extractVideoId(url); + if ( + dirId && + (await isDeferredAuthExcluded( + path.join(dataDir, dirId), + runCookiePolicy, + )) + ) { + deferredAuthCount++; + continue; + } newUrls.push(url); } opts.onLog( `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`, ); + if (deferredAuthCount > 0) { + opts.onLog( + `Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`, + ); + } if (newUrls.length > 0) { - const res = await runManagedDownloads(opts, newUrls, opts.channelConfig); + const res = await runManagedDownloads( + opts, + newUrls, + opts.channelConfig, + runCookiePolicy, + ); totalNew += res.okCount; totalFailed += res.failedCount; totalSkipped += res.skippedCount; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +- **Channel groups gain a "Show channels as individual chips" checkbox on the Site form.** Toggling it on marks the group `inline` in site.json, so the viewer renders each member channel as its own loose chip in the filter row instead of a collapsible group box. New sites' seeded default group ("All channels") ships with it **on**; groups added manually on the Site form or created inline from a channel form default **off**. See `editor/app/sites/components/SiteForm.tsx`, `common/lib/channelGroups.ts`, and `editor/e2e/sites-crud.spec.ts`. +- **Cookies-from-browser is now configurable: three cookie modes, per-channel overrides, and a "Needs cookies" bucket.** The single global cookies value used to be hard-wired to one behavior (passed only on the download auth-retry attempt). Settings now carry a **Cookie mode** next to the value — **When required** (default; the old behavior, now also cookie-retrying a failed *metadata prefetch*, so an age-gated video succeeds on its first real attempt and reports `ok-with-cookies` with a `metadata-prefetch-auth-retry` attempt in `download-outcome.json`), **Always** (every yt-dlp invocation carries `--cookies-from-browser`: enumeration, prefetch, availability probes, downloads — for channels whose listing itself needs auth; a clean success still reports plain `ok`), and **Defer** (normal runs never use cookies: known `needs_auth` videos are excluded from sync/download-missing batches — previously they were re-attempted and re-failed on *every* run — and wait for a manual cookie run). Both the value and the mode can be overridden per channel in the channel form's Advanced section (blank/Inherit = use global). Every channel's Download stage gains a **Needs cookies (N)** card listing undownloaded videos whose recorded availability is cookie-recoverable (`needs_auth`, and also `members_only`/`private`, which batches always excluded but a subscribed/owning account's cookies can fetch) with a **Download with cookies** button that re-runs them with cookies forced on every invocation (bookmarkable, like the other bucket jobs; it warns if no cookie value is configured anywhere). Caveats: in defer mode a *manual single-video* download intentionally gets no cookies and no auth retry — the bucket button is the explicit cookie path; availability probes only attach cookies in Always mode, so needs_auth keeps being observed (defer depends on that signal). Back-compat: an existing `settings.json` without `cookieMode` behaves exactly as before (`when-required`). See `common/lib/cookiePolicy.ts`, `common/ytdlp/{downloadOneManaged,runYtdlp}.ts`, `common/controller/{channelSnapshot,checkAvailability}.ts`, and `editor/e2e/cookies-mode.spec.ts`. +- **Channel forms now edit site membership directly — pick sites, groups, and create groups inline.** A channel's site membership used to be editable only from the Site form (`/sites/<id>`), and creating a channel silently appended it to the active site (a hidden `activeSite` field) with no group choice and no visibility. Both the **New channel** and channel **Configure** forms now carry a **Sites** section: every configured site listed with a membership checkbox and a compact group dropdown — `(default)`, any existing group, or **"+ New group…"**, which reveals a name input and creates the group on that site (id slugified from the name, reused if it already exists) as part of the save. On create, the active site (`?site=`) is pre-checked, reproducing the old behavior but visibly and overridably; on edit, current memberships pre-check with their groups, and unchecking removes the membership. The checked sites serialize into one hidden `siteMembershipsJson` field (the SiteForm hidden-JSON precedent); the server plans all writes up front — validating group ids, preserving other channels' entries and this channel's `order`, erroring clearly on a since-deleted group, skipping since-deleted sites, and rewriting only sites that actually changed. See the new `editor/app/channels/{lib/siteMemberships.ts,components/SiteMembershipsSection.tsx}`, `editor/app/channels/{actions.ts,components/{ChannelForm,ChannelFormClient}.tsx,new/page.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-site-membership.spec.ts`. +- **The monitor widget can now start a sync and shows sync freshness + scheduler health — each behind its own flag.** The widget was start-a-sync-less and said nothing about how fresh your channels were. Five new opt-in URL flags, all toggleable in the builder and the in-widget gear (all default off, so existing links are unchanged): **(1) a channel-aware Sync button** (`sync=1`) — pinned to one channel (`channel=X`) it runs that channel's streaming sync; otherwise it sweeps all channels (`syncAllChannelsAction`), briefly showing `Queued X · skipped Y`. **(2) A last-sync readout** (`lastsync=1`) — `Last full sync: …` from a new persisted `lastSyncAllAt` marker written at the end of each Sync-all sweep, plus a `Last channel sync: …` line when an individual channel synced more recently. **(3) A scheduler-status strip** (`sched=1`) — `Auto-sync on/off · next … · last run …`, derived from the same `buildScheduleView` the `/scheduler` page uses. **(4) An absolute-time toggle** (`abstime=1`) — readouts show locale timestamps instead of relative "5m ago". **(5) A Sync-all confirm** (`syncask=1`) — a `window.confirm` before a full sweep. The two readouts poll a new lightweight `/api/widget/sync` route (a few scalars, 15s floor) and, with the Sync button/controls, keep the widget from collapsing to "Idle" — sync health is exactly what you check when nothing's running. See `editor/app/widget/lib/config.ts`, `editor/app/widget/components/{MonitorWidget,WidgetControls,WidgetConfigForm}.tsx`, the new `editor/app/api/widget/sync/route.ts`, `common/jobs/syncSchedulerState.ts` (`lastSyncAllAt`), `editor/app/channels/actions.ts`, and `editor/e2e/widget.spec.ts`. - **The audio-integrity check now tightens its probe interval when a source starts serving corruption, then relaxes as it stabilises.** Previously the integrity probe ran on a fixed cadence (default 60s) for the whole download, so up to ~60s of bytes were downloaded — and discarded — between a corruption event and the checkpoint that caught it. The interval is now adaptive (AIMD, like TCP congestion control, inverted): each **malformed** checkpoint **halves** the live interval (60→30→15→10s, floored at the existing `AUDIO_CHECK_INTERVAL_MIN_SECONDS` of 10s), so a misbehaving source gets probed more aggressively and wastes fewer bytes per rollback; a run of clean checkpoints then **steps it back up** additively (+15s after every 2 clean probes) toward the configured interval. The reduced cadence persists across yt-dlp relaunches for the rest of the download run. Fully backward compatible — a clean download never leaves the configured interval. Tunable via constants in `common/lib/channelConfig.ts` (`AUDIO_CHECK_INTERVAL_BACKOFF_FACTOR_DEFAULT`, `AUDIO_CHECK_INTERVAL_RECOVER_STEP_SECONDS`, `AUDIO_CHECK_INTERVAL_RECOVER_AFTER_CLEAN`) plus test-only env overrides. See `common/ytdlp/audioCheckCadence.ts` (pure AIMD math + `audioCheckCadence.test.ts`) and `common/ytdlp/audioCheckedDownload.ts` (`resolveKnobs`, the watcher loop, and the advance/malformed checkpoint branches). - **Docker build mode is now real: build every site in parallel, then deploy them serially.** The `Docker` build mode (Settings → Build pipeline) was previously a stub that fell back to the basic build. It now runs a proper pipeline, driven by a new **Build all sites** control on the Deploy page (one job, one log, one Cancel). The shared, corpus-scale work — the search index, the per-site staging, and the downloadable archive zips — runs **once on the host**; then each site's `compose + next build` runs in its **own container in parallel** (capped by the **Max parallel builds** setting), each writing an isolated per-site `out/` under `export/.export-builds/<siteId>/`; then the built sites **deploy serially** on the host (R2 upload + `wrangler pages deploy`), tolerant of a single site failing. Containers are read-only over the shared corpus/index/archive cache and run as your host user so outputs aren't root-owned. The image (`Dockerfile.build`, tag from **Build image**) is built/reused via Docker layer caching; when no container engine is available the action falls back to a serial host build+deploy. New env knobs: `DOCKER_BIN` (e.g. `podman`), `DOCKER_BUILD_MEMORY`/`DOCKER_BUILD_CPUS` (per-container caps). See `editor/app/deploy/buildDeployCore.ts` (`runDockerBuildAllPhase`/`runDockerDeployAllPhase`), `editor/app/build/buildAction.ts` (`buildAndDeployAllSitesAction`/`buildAllSitesAction`), `editor/app/deploy/components/BuildAllSitesButton.tsx`, `Dockerfile.build`, `docker/build-site.sh`, `common/bin/build-archives.ts`, and **[DEPLOY_DOCKER.md](../DEPLOY_DOCKER.md)**. diff --git a/editor/app/api/widget/sync/route.ts b/editor/app/api/widget/sync/route.ts @@ -0,0 +1,72 @@ +import { NextResponse } from "next/server"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { listChannels } from "yt-dlp-transcript-common/controller/channels"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { buildScheduleView } from "yt-dlp-transcript-common/jobs/syncScheduler"; +import { readSchedulerState } from "yt-dlp-transcript-common/jobs/syncSchedulerState"; + +export const dynamic = "force-dynamic"; + +// Backs the monitor widget's last-sync and scheduler-status strips. Keeps the +// payload tiny (a handful of scalars) rather than shipping the full +// scheduler-status payload on every poll — the widget only needs freshness +// markers and a one-line scheduler summary. Reuses the same helpers the +// /scheduler status page does so the two never drift. +export type WidgetSyncPayload = { + // Epoch ms of the last manual full "Sync all" sweep. null = never. + lastSyncAllAt: number | null; + // Newest per-channel lastSyncedAt across all channels. null = none synced. + lastIndividualSyncAt: number | null; + scheduler: { + enabled: boolean; + lastRunAt: number | null; // last scheduled sweep (state.runs[0].at) + nextRunAt: number | null; // soonest eligible channel's nextDueAt + overdue: boolean; // some eligible channel is due now + }; +}; + +export async function GET() { + const paths = getPaths(); + const settings = getSettings(); + const state = await readSchedulerState(paths); + const now = Date.now(); + const channels = await listChannels(paths); + + let lastIndividualSyncAt: number | null = null; + for (const c of channels) { + const last = c.config.lastSyncedAt; + if (!last) continue; + const parsed = Date.parse(last); + if (Number.isNaN(parsed)) continue; + if (lastIndividualSyncAt === null || parsed > lastIndividualSyncAt) { + lastIndividualSyncAt = parsed; + } + } + + const view = buildScheduleView({ + channels: channels.map((c) => ({ slug: c.slug, config: c.config })), + scheduler: settings.syncScheduler, + state, + now, + }); + const eligible = view.filter((v) => v.autoSyncEligible); + let nextRunAt: number | null = null; + let overdue = false; + for (const v of eligible) { + if (v.overdue) overdue = true; + if (v.nextDueAt !== null && (nextRunAt === null || v.nextDueAt < nextRunAt)) { + nextRunAt = v.nextDueAt; + } + } + + return NextResponse.json({ + lastSyncAllAt: state.lastSyncAllAt, + lastIndividualSyncAt, + scheduler: { + enabled: settings.syncScheduler.enabled, + lastRunAt: state.runs[0]?.at ?? null, + nextRunAt, + overdue, + }, + } satisfies WidgetSyncPayload); +} diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx @@ -17,6 +17,9 @@ type Props = { // The snapshot bucket these ids represent. When set, the launched retry job // is bookmarkable and re-runs against the current bucket members. bucketKey?: ReplayBucket; + // Needs-cookies bucket: force cookie mode "always" for the run and label the + // button "Download with cookies" so the manual cookie path is explicit. + forceCookies?: boolean; }; export function RetryBucketControl({ @@ -26,6 +29,7 @@ export function RetryBucketControl({ defaultQueueKey, existingQueues, bucketKey, + forceCookies, }: Props) { const [queue, setQueue] = useState(defaultQueueKey); const [handlingOverride, setHandlingOverride] = useState(""); @@ -47,10 +51,15 @@ export function RetryBucketControl({ abortOnError, handlingOverride || undefined, bucketKey, + forceCookies, ) } cancelAction={cancelJobAction} - buttonLabel={`Retry (${ids.length})`} + buttonLabel={ + forceCookies + ? `Download with cookies (${ids.length})` + : `Retry (${ids.length})` + } runningLabel="Retrying…" label={`Retry ${actionLabel}`} extraControls={ diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx @@ -28,6 +28,7 @@ type Props = { excludedFromDownload: ExcludedFromDownload; noTranscriptIds: string[]; partialDownloadIds: string[]; + needsCookiesIds: string[]; missingShard: ShardConfigSummary | null; }; @@ -40,6 +41,7 @@ export function DownloadStage({ excludedFromDownload, noTranscriptIds, partialDownloadIds, + needsCookiesIds, missingShard, }: Props) { const [downloadQueue, setDownloadQueue] = useState(defaultQueueKey); @@ -217,6 +219,12 @@ export function DownloadStage({ defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} /> + <NeedsCookiesList + slug={slug} + ids={needsCookiesIds} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> </div> ); } @@ -416,6 +424,54 @@ function PartialDownloadsList({ ); } +function NeedsCookiesList({ + slug, + ids, + defaultQueueKey, + existingQueues, +}: { + slug: string; + ids: string[]; + defaultQueueKey: string; + existingQueues: string[]; +}) { + if (ids.length === 0) return null; + return ( + <div className="flex flex-col gap-2 rounded border border-border p-3"> + <div> + <h4 className="text-sm font-semibold"> + Needs cookies ({ids.length}) + </h4> + <p className="text-xs text-muted-foreground"> + Undownloaded videos that yt-dlp reported as auth-gated + (age-restricted, members-only, or private) — browser cookies from a + signed-in account may recover them. The button below re-runs them + with <code>--cookies-from-browser</code> forced on every invocation, + using the cookie value from Settings (or this channel&apos;s + override). + </p> + </div> + <VideoIdList + slug={slug} + ids={ids} + ariaLabel="needs cookies list" + emptyAriaLabel="needs cookies empty" + emptyMessage="None" + itemAriaLabel={(id) => `needs cookies ${id}`} + /> + <RetryBucketControl + slug={slug} + ids={ids} + actionLabel="needs cookies" + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + bucketKey="needsCookies" + forceCookies + /> + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -30,6 +30,7 @@ export function normalizeBuckets( skippedByFilter: raw?.skippedByFilter ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], shortAudio: raw?.shortAudio ?? [], + needsCookies: raw?.needsCookies ?? [], }; } diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -35,7 +35,13 @@ import { TRANSCRIPTION_QUEUE, } from "yt-dlp-transcript-common/lib/platform"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; +import { listSites } from "yt-dlp-transcript-common/lib/site"; +import { sortGroups } from "yt-dlp-transcript-common/lib/channelGroups"; import { ChannelFormClient } from "../components/ChannelFormClient"; +import type { + InitialMembership, + SiteMembershipOption, +} from "../components/SiteMembershipsSection"; import { DeleteChannelForm } from "../components/DeleteChannelForm"; import { RenameChannelForm } from "../components/RenameChannelForm"; import { RunningJobsList } from "../../jobs/components/RunningJobsList"; @@ -219,6 +225,25 @@ export default async function ChannelDetailPage({ "danger", ]; + // Sites membership section of the configure form: every configured site plus + // this channel's current membership (and group) on each. + const allSites = listSites(paths); + const siteOptions: SiteMembershipOption[] = allSites.map((s) => ({ + siteId: s.siteId, + siteTitle: s.siteTitle, + defaultGroupId: s.defaultGroupId, + groups: sortGroups(s.groups).map((g) => ({ id: g.id, name: g.name })), + })); + const initialMemberships: InitialMembership[] = allSites.flatMap((s) => { + const m = s.channels.find((c) => c.slug === slug); + if (!m) return []; + return [ + m.groupId + ? { siteId: s.siteId, groupId: m.groupId } + : { siteId: s.siteId }, + ]; + }); + const panels: Partial<Record<StageId, ReactNode>> = { configure: ( <ChannelFormClient @@ -230,6 +255,8 @@ export default async function ChannelDetailPage({ } initial={{ slug, config }} submitLabel="Save changes" + sites={siteOptions} + initialMemberships={initialMemberships} /> ), playlist: ( @@ -250,6 +277,7 @@ export default async function ChannelDetailPage({ excludedFromDownload={excludedFromDownload} noTranscriptIds={actionableNoTranscriptIds} partialDownloadIds={buckets.partialDownloads} + needsCookiesIds={buckets.needsCookies} missingShard={downloadMissingShard} /> ), diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -24,6 +24,7 @@ import { import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { @@ -70,6 +71,10 @@ async function runPipelineAction( shardIndex?: number; bucketIds?: ReadonlyArray<string>; handlingOverride?: ChannelHandling; + // retry-bucket only (the "Needs cookies" bucket): force cookie mode + // "always" for this run so every yt-dlp invocation carries the configured + // cookies and the defer-mode exclusion is bypassed. + forceCookies?: boolean; // Per-run persistence overrides (Phase 2), forwarded to runYtdlp → // downloadOneManaged for every managed download in this run. keepSourceVideoOverride?: boolean; @@ -166,6 +171,7 @@ async function runPipelineAction( shardIndex: options?.shardIndex, bucketIds: options?.bucketIds, handlingOverride: options?.handlingOverride, + forceCookies: options?.forceCookies, keepSourceVideoOverride: options?.keepSourceVideoOverride, extractImmediately: options?.extractImmediately, audioFormatOverride: options?.audioFormatOverride, @@ -302,6 +308,8 @@ export async function retryBucketAction( // against the CURRENT bucket; omitted by ad-hoc checkbox selections, which are // therefore not bookmarkable. bucketKey?: ReplayBucket, + // Needs-cookies bucket only: run with cookie mode forced to "always". + forceCookies?: boolean, ): Promise<StreamActionResult> { if (!Array.isArray(bucketIds) || bucketIds.length === 0) { return { ok: false, error: "No video IDs supplied for retry." }; @@ -321,7 +329,7 @@ export async function retryBucketAction( kind: "retry-bucket", slug, bucket: bucketKey, - params: { queueKey, abortOnError, handlingOverride }, + params: { queueKey, abortOnError, handlingOverride, forceCookies }, } : undefined; return runPipelineAction( @@ -329,7 +337,13 @@ export async function retryBucketAction( "retry-bucket", "retry-bucket", queueKey, - { bucketIds, handlingOverride: handling, abortOnError, spec }, + { + bucketIds, + handlingOverride: handling, + abortOnError, + forceCookies, + spec, + }, ); } @@ -380,7 +394,7 @@ export async function importVideoAction( videoUrl, onLog: task.onLog, signal, - globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + cookiePolicy: resolveCookiePolicy(settings, channelConfig), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -36,6 +36,7 @@ import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transc import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos"; import { unpersistSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { @@ -172,7 +173,7 @@ export async function downloadVideoPipelineAction( videoUrl: url, onLog: task.onLog, signal, - globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + cookiePolicy: resolveCookiePolicy(settings, r.config), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, downloadFormatPreset: resolveDownloadFormatPreset({ @@ -237,7 +238,7 @@ export async function redownloadToArchiveAction( videoUrl: url, onLog: task.onLog, signal, - globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + cookiePolicy: resolveCookiePolicy(settings, r.config), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts @@ -17,17 +17,21 @@ import { renameChannel } from "yt-dlp-transcript-common/controller/renameChannel import { generateChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; -import { - getSite, - isValidSiteId, - listSiteIds, - writeSite, -} from "yt-dlp-transcript-common/lib/site"; +import type { Site } from "yt-dlp-transcript-common/lib/site"; import { activeSyncSlugs } from "yt-dlp-transcript-common/jobs/syncJobs"; import { + readSchedulerState, + writeSchedulerState, +} from "yt-dlp-transcript-common/jobs/syncSchedulerState"; +import { CHANNEL_FORM_FIELDS, parseChannelForm, } from "./components/parseChannelForm"; +import { + applySiteWrites, + parseSiteMembershipsField, + planSiteMembershipWrites, +} from "./lib/siteMemberships"; import { syncAction } from "./[slug]/pipelineActions"; export type ActionResult = { error: string } | undefined; @@ -53,28 +57,35 @@ export async function createChannelAction( }; } const paths = getPaths(); + // Validate + plan the site-membership writes from the form's Sites section + // BEFORE the channel exists, so a rejected submit can be corrected and + // resubmitted without hitting "already exists". A null parse result (field + // absent — zero sites configured) skips membership work entirely. + let siteWrites: Site[] = []; + try { + const requests = parseSiteMembershipsField( + formData.get("siteMembershipsJson"), + ); + if (requests) { + siteWrites = planSiteMembershipWrites(paths, slug, requests); + } + } catch (e) { + return { error: (e as Error).message }; + } if (await channelExists(paths, slug)) { return { error: `Channel "${slug}" already exists` }; } await createChannel(paths, slug, config); - // When created under a specific active site, add the new channel to that - // site's membership so it's visible in the scoped Channels list. No-op under - // "all sites" (the hidden field is omitted by the form in that case). - const activeSite = formData.get("activeSite"); - if ( - typeof activeSite === "string" && - isValidSiteId(activeSite) && - listSiteIds(paths).includes(activeSite) - ) { - const site = getSite(activeSite, paths); - if (!site.channels.some((c) => c.slug === slug)) { - await writeSite( - { ...site, channels: [...site.channels, { slug }] }, - paths, - ); - } + try { + await applySiteWrites(siteWrites, paths); + } catch (e) { + // The channel itself was created; don't redirect as if nothing happened. + return { + error: `Channel "${slug}" was created, but updating site memberships failed: ${(e as Error).message}. Open its Configure panel to retry.`, + }; } revalidatePath("/channels"); + revalidatePath("/sites"); revalidatePath("/"); redirect(`/channels/${slug}`); } @@ -93,6 +104,20 @@ export async function updateChannelAction( const paths = getPaths(); const existing = await readChannelConfig(paths, slug); if (!existing) return { error: `Channel "${slug}" not found` }; + // Plan the site-membership writes (Sites section) before touching anything, + // so bad input errors out with no partial write. Null = field absent + // (zero sites configured / legacy submit) → leave memberships alone. + let siteWrites: Site[] = []; + try { + const requests = parseSiteMembershipsField( + formData.get("siteMembershipsJson"), + ); + if (requests) { + siteWrites = planSiteMembershipWrites(paths, slug, requests); + } + } catch (e) { + return { error: (e as Error).message }; + } // The form parser only emits keys whose form value is meaningful, so a // cleared input is absent from `parsed.config`. A plain spread would keep // the stale value from `existing`. Clear every form-managed key from the @@ -103,11 +128,18 @@ export async function updateChannelAction( for (const key of CHANNEL_FORM_FIELDS) delete merged[key]; Object.assign(merged, parsed.config); await writeChannelConfig(paths, slug, merged); + try { + await applySiteWrites(siteWrites, paths); + } catch (e) { + return { error: (e as Error).message }; + } // Config changes (e.g. audioFormat / handling) feed snapshot buckets, so // refresh the report through the global debounced scheduler. requestChannelSnapshot(paths, slug); revalidatePath("/channels"); revalidatePath(`/channels/${slug}`); + revalidatePath("/sites"); + revalidatePath("/"); return undefined; } @@ -223,6 +255,12 @@ export async function syncAllChannelsAction(): Promise<SyncAllResult> { // runManagedFunction's onLog regardless. void result.stream.cancel(); } + // Record the sweep's freshness marker for the monitor widget's last-sync + // readout. Read-modify-write right before the write keeps the clobber window + // vs. a concurrent scheduler tick minimal (single-user editor — acceptable). + const state = await readSchedulerState(paths); + state.lastSyncAllAt = Date.now(); + await writeSchedulerState(paths, state); return { queued, skipped }; } diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx @@ -15,15 +15,23 @@ import { AUDIO_CHECK_MAX_ROLLBACKS_MIN, } from "yt-dlp-transcript-common/lib/channelConfig"; import { SYNC_INTERVAL_PRESETS } from "../../scheduler/intervalPresets"; +import { + SiteMembershipsSection, + type InitialMembership, + type SiteMembershipOption, +} from "./SiteMembershipsSection"; type Props = { action: string | ((formData: FormData) => void | Promise<void>); initial?: { slug: string; config: ChannelConfig }; submitLabel: string; errorMessage?: string; - // Extra hidden fields submitted with the form (e.g. the active site so a newly - // created channel can be added to that site's membership). - hiddenFields?: Record<string, string>; + // All configured sites, for the Sites membership section. + sites: SiteMembershipOption[]; + // Edit mode: the channel's current site memberships. + initialMemberships?: InitialMembership[]; + // Create mode: the active site (from ?site=) to pre-check. + activeSiteId?: string | null; }; export function ChannelForm({ @@ -31,16 +39,14 @@ export function ChannelForm({ initial, submitLabel, errorMessage, - hiddenFields, + sites, + initialMemberships, + activeSiteId, }: Props) { const isEdit = !!initial; const c = initial?.config; return ( <form action={action} className="flex flex-col gap-4 max-w-xl"> - {hiddenFields && - Object.entries(hiddenFields).map(([name, value]) => ( - <input key={name} type="hidden" name={name} defaultValue={value} /> - ))} {errorMessage && ( <div className="rounded border border-destructive/30 bg-destructive-soft px-3 py-2 text-sm text-destructive"> {errorMessage} @@ -69,13 +75,13 @@ export function ChannelForm({ placeholder="kebab-case (auto-derived from name if blank)" /> )} - <p className="text-xs text-muted-foreground"> - Channel grouping is now configured per site under{" "} - <a href="/sites" className="underline"> - Sites - </a> - , where each site picks its own channels and their groups. - </p> + </Section> + <Section title="Sites"> + <SiteMembershipsSection + sites={sites} + initialMemberships={initialMemberships} + activeSiteId={activeSiteId} + /> </Section> <Section title="Source"> <fieldset className="flex flex-col gap-2"> @@ -245,6 +251,39 @@ export function ChannelForm({ className="rounded border border-border bg-card px-2 py-1 text-sm font-mono" /> </label> + <Field + label="Cookies from browser" + name="cookiesFromBrowser" + defaultValue={c?.cookiesFromBrowser ?? ""} + placeholder="(inherit global)" + hint="Per-channel override of the global yt-dlp --cookies-from-browser browser spec (e.g. firefox, chrome:Default). Blank inherits the global value from /settings." + /> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Cookie mode</span> + <select + name="cookieMode" + defaultValue={c?.cookieMode ?? ""} + aria-label="cookie mode" + className="rounded border border-border bg-card px-2 py-1 text-sm" + > + <option value="">Inherit global</option> + <option value="when-required"> + When required — retry auth/age failures with cookies + </option> + <option value="always"> + Always — pass cookies on every yt-dlp invocation + </option> + <option value="defer"> + Defer — never in normal runs; collect into "Needs cookies" + </option> + </select> + <span className="text-xs text-muted-foreground"> + When this channel&apos;s downloads use browser cookies. Leave on + Inherit to use the global mode from /settings. Defer excludes + auth-gated videos from batch runs and gathers them into the + &ldquo;Needs cookies&rdquo; bucket on the Download stage. + </span> + </label> <label className="flex flex-col gap-1 text-sm"> <span className="font-medium"> Sleep between managed downloads diff --git a/editor/app/channels/components/ChannelFormClient.tsx b/editor/app/channels/components/ChannelFormClient.tsx @@ -5,6 +5,10 @@ import { ChannelForm } from "./ChannelForm"; import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import type { ActionResult } from "../actions"; import { ALL_SITES } from "../../lib/activeSite"; +import type { + InitialMembership, + SiteMembershipOption, +} from "./SiteMembershipsSection"; type Props = { action: ( @@ -13,16 +17,26 @@ type Props = { ) => Promise<ActionResult>; initial?: { slug: string; config: ChannelConfig }; submitLabel: string; + sites: SiteMembershipOption[]; + initialMemberships?: InitialMembership[]; }; -export function ChannelFormClient({ action, initial, submitLabel }: Props) { +export function ChannelFormClient({ + action, + initial, + submitLabel, + sites, + initialMemberships, +}: Props) { const [state, formAction] = useActionState<ActionResult, FormData>( action, undefined, ); // The active site is mirrored into the `?site=` URL param by SiteScopeSelect. // Read it here (avoiding useSearchParams so this form needn't be wrapped in a - // Suspense boundary) so a newly created channel joins that site's membership. + // Suspense boundary) so a newly created channel pre-checks that site in the + // Sites membership section. Only meaningful in create mode — an existing + // channel's memberships come from initialMemberships. const [activeSite, setActiveSite] = useState<string | null>(null); useEffect(() => { try { @@ -31,15 +45,17 @@ export function ChannelFormClient({ action, initial, submitLabel }: Props) { /* ignore */ } }, []); - const hiddenFields = - activeSite && activeSite !== ALL_SITES ? { activeSite } : undefined; + const activeSiteId = + !initial && activeSite && activeSite !== ALL_SITES ? activeSite : null; return ( <ChannelForm action={formAction} initial={initial} submitLabel={submitLabel} errorMessage={state?.error} - hiddenFields={hiddenFields} + sites={sites} + initialMemberships={initialMemberships} + activeSiteId={activeSiteId} /> ); } diff --git a/editor/app/channels/components/DeleteChannelForm.tsx b/editor/app/channels/components/DeleteChannelForm.tsx @@ -27,6 +27,7 @@ export function DeleteChannelForm({ slug, action }: Props) { name="confirmSlug" required placeholder={slug} + aria-label="confirm slug to delete" className="rounded border border-border bg-card px-2 py-1 text-sm font-mono" /> <button diff --git a/editor/app/channels/components/SiteMembershipsSection.tsx b/editor/app/channels/components/SiteMembershipsSection.tsx @@ -0,0 +1,192 @@ +"use client"; + +import { useEffect, useState } from "react"; + +// The channel form's per-site membership picker: every configured site with a +// checkbox (member or not) plus a group dropdown, including a "+ New group…" +// option that reveals a name input. Serializes the checked sites into the +// hidden `siteMembershipsJson` field consumed by lib/siteMemberships.ts — +// same hidden-JSON-field pattern as SiteForm's groupsJson/channelsJson. + +export type SiteMembershipOption = { + siteId: string; + siteTitle: string; + defaultGroupId: string; + // Pre-sorted server-side via sortGroups. + groups: { id: string; name: string }[]; +}; + +export type InitialMembership = { siteId: string; groupId?: string }; + +// Sentinel select value for "+ New group…". Can never collide with a real +// group id — underscores fail isValidGroupId (same trick as ALL_SITES). +const NEW_GROUP = "__new__"; + +type Row = { groupId: string; newGroupName: string }; + +type Props = { + sites: SiteMembershipOption[]; + // Edit mode: the channel's current memberships (pre-checked). + initialMemberships?: InitialMembership[]; + // Create mode: the active site from `?site=` to pre-check. Arrives via a + // mount effect in ChannelFormClient, before any user interaction. + activeSiteId?: string | null; +}; + +export function SiteMembershipsSection({ + sites, + initialMemberships, + activeSiteId, +}: Props) { + // Presence in the map = checked. groupId "" = the site's default group. + const [selected, setSelected] = useState<Map<string, Row>>( + () => + new Map( + (initialMemberships ?? []).map((m) => [ + m.siteId, + { groupId: m.groupId ?? "", newGroupName: "" }, + ]), + ), + ); + + // Create-mode pre-check of the active site (mirrors the old hidden + // `activeSite` field's behavior). Fires before user interaction, so no + // clobber guard beyond "already checked" is needed. + useEffect(() => { + if (!activeSiteId || !sites.some((s) => s.siteId === activeSiteId)) return; + setSelected((prev) => { + if (prev.has(activeSiteId)) return prev; + const next = new Map(prev); + next.set(activeSiteId, { groupId: "", newGroupName: "" }); + return next; + }); + }, [activeSiteId, sites]); + + if (sites.length === 0) { + // No hidden field at all: the actions skip membership reconciliation. + return ( + <p className="text-xs text-muted-foreground"> + No sites configured — create one under{" "} + <a href="/sites" className="underline"> + Sites + </a> + . + </p> + ); + } + + const toggle = (siteId: string, on: boolean) => + setSelected((prev) => { + const next = new Map(prev); + if (on) next.set(siteId, prev.get(siteId) ?? { groupId: "", newGroupName: "" }); + else next.delete(siteId); + return next; + }); + const setGroup = (siteId: string, groupId: string) => + setSelected((prev) => { + const next = new Map(prev); + const row = prev.get(siteId) ?? { groupId: "", newGroupName: "" }; + next.set(siteId, { ...row, groupId }); + return next; + }); + const setNewGroupName = (siteId: string, newGroupName: string) => + setSelected((prev) => { + const next = new Map(prev); + const row = prev.get(siteId) ?? { groupId: "", newGroupName: "" }; + next.set(siteId, { ...row, newGroupName }); + return next; + }); + + // One entry per CHECKED site; an unchecked site is simply absent (= remove + // membership on edit). See MembershipRequest in lib/siteMemberships.ts. + const membershipsJson = JSON.stringify( + sites + .filter((s) => selected.has(s.siteId)) + .map((s) => { + const row = selected.get(s.siteId)!; + if (row.groupId === NEW_GROUP) { + const name = row.newGroupName.trim(); + // The visible name input is `required`, so an empty name never + // submits; falling back to default is belt-and-braces only. + return name + ? { siteId: s.siteId, newGroupName: name } + : { siteId: s.siteId }; + } + return row.groupId + ? { siteId: s.siteId, groupId: row.groupId } + : { siteId: s.siteId }; + }), + ); + + return ( + <div className="flex flex-col gap-1"> + <input type="hidden" name="siteMembershipsJson" value={membershipsJson} /> + <p className="text-xs text-muted-foreground"> + Pick the sites this channel appears on, and the group it belongs to on + each. + </p> + {sites.map((s) => { + const row = selected.get(s.siteId); + const on = !!row; + const label = s.siteTitle || s.siteId; + return ( + <div + key={s.siteId} + className="flex flex-col gap-1 py-1 border-b border-border last:border-b-0" + > + <div className="flex items-center gap-2 text-sm"> + <label className="flex items-center gap-2 flex-1 min-w-0"> + <input + type="checkbox" + checked={on} + onChange={(e) => toggle(s.siteId, e.target.checked)} + aria-label={`Include on ${label}`} + /> + <span className="truncate"> + {label}{" "} + <code className="text-xs text-muted-foreground"> + {s.siteId} + </code> + </span> + </label> + {on && ( + <select + value={row.groupId} + onChange={(e) => setGroup(s.siteId, e.target.value)} + aria-label={`Group for ${label}`} + className="rounded border border-border bg-card px-1.5 py-0.5 text-xs shrink-0" + > + <option value="">(default)</option> + {s.groups.map((g) => ( + <option key={g.id} value={g.id}> + {g.name || g.id} + </option> + ))} + <option value={NEW_GROUP}>+ New group…</option> + </select> + )} + </div> + {on && row.groupId === NEW_GROUP && ( + <input + type="text" + value={row.newGroupName} + onChange={(e) => setNewGroupName(s.siteId, e.target.value)} + required + placeholder="New group name" + aria-label={`New group name for ${label}`} + className="ml-6 rounded border border-border bg-card px-2 py-1 text-sm" + /> + )} + </div> + ); + })} + <p className="text-xs text-muted-foreground"> + Full group management (rename, reorder, defaults) lives under{" "} + <a href="/sites" className="underline"> + Sites + </a> + . + </p> + </div> + ); +} diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts @@ -17,6 +17,7 @@ import { detectPlatform, type Platform, } from "yt-dlp-transcript-common/lib/platform"; +import { isCookieMode } from "yt-dlp-transcript-common/lib/cookiePolicy"; import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; export type ParsedChannelForm = { @@ -41,6 +42,8 @@ export const CHANNEL_FORM_FIELDS = [ "ytdlpExtraArgs", "syncIntervalMinutes", "sleepBetweenDownloadsSeconds", + "cookiesFromBrowser", + "cookieMode", "audioCheck", ] as const satisfies ReadonlyArray<keyof ChannelConfig>; @@ -143,6 +146,12 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm { syncIntervalMinutes = n; } + // Cookie overrides. Blank value / "Inherit global" mode = omit, so a cleared + // input actually clears the stored override (see CHANNEL_FORM_FIELDS). + const cookiesFromBrowser = stringOrUndef(formData, "cookiesFromBrowser"); + const cookieModeRaw = String(formData.get("cookieMode") ?? "").trim(); + const cookieMode = isCookieMode(cookieModeRaw) ? cookieModeRaw : undefined; + const audioCheckEnabled = formData.get("audioCheckEnabled") != null; let audioCheck: AudioCheckConfig | undefined; if (audioCheckEnabled) { @@ -221,6 +230,8 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm { if (sleepBetweenDownloadsSeconds != null) { config.sleepBetweenDownloadsSeconds = sleepBetweenDownloadsSeconds; } + if (cookiesFromBrowser) config.cookiesFromBrowser = cookiesFromBrowser; + if (cookieMode) config.cookieMode = cookieMode; if (audioCheck) config.audioCheck = audioCheck; return { name, slug, config }; diff --git a/editor/app/channels/lib/siteMemberships.ts b/editor/app/channels/lib/siteMemberships.ts @@ -0,0 +1,152 @@ +import slugify from "@sindresorhus/slugify"; +import { isValidGroupId } from "yt-dlp-transcript-common/lib/channelGroups"; +import type { Paths } from "yt-dlp-transcript-common/lib/paths"; +import { + getSite, + listSiteIds, + writeSite, + type Site, + type SiteChannelMembership, +} from "yt-dlp-transcript-common/lib/site"; + +// Server-side half of the channel form's "Sites" membership section (see +// SiteMembershipsSection.tsx). The form serializes one entry per CHECKED site +// into the hidden `siteMembershipsJson` field; a site absent from the array +// means "not a member" (removal on edit). The field being absent entirely +// (zero sites configured, or a legacy submit) means "don't touch memberships". + +export type MembershipRequest = { + siteId: string; + // Explicit existing group on that site; absent = the site's default group. + groupId?: string; + // Mutually exclusive with groupId: create this group (id = slugify(name)) + // on the site and assign the channel to it. + newGroupName?: string; +}; + +// Parse the raw hidden-field value. Returns null when the field is absent so +// callers can skip membership reconciliation entirely; throws on a payload +// that isn't a JSON array (never produced by our form). +export function parseSiteMembershipsField( + raw: FormDataEntryValue | null, +): MembershipRequest[] | null { + if (typeof raw !== "string") return null; + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + throw new Error("Malformed site memberships payload"); + } + if (!Array.isArray(parsed)) { + throw new Error("Malformed site memberships payload"); + } + const out: MembershipRequest[] = []; + const seen = new Set<string>(); + for (const entry of parsed) { + if (!entry || typeof entry !== "object") continue; + const r = entry as Record<string, unknown>; + if (typeof r.siteId !== "string" || !r.siteId || seen.has(r.siteId)) { + continue; + } + seen.add(r.siteId); + const req: MembershipRequest = { siteId: r.siteId }; + // newGroupName wins when both are present (the form never emits both). + if (typeof r.newGroupName === "string" && r.newGroupName.trim()) { + req.newGroupName = r.newGroupName.trim(); + } else if (typeof r.groupId === "string" && r.groupId.trim()) { + req.groupId = r.groupId.trim(); + } + out.push(req); + } + return out; +} + +// Compute the Site objects that need rewriting so `slug`'s membership matches +// `requests`. Pure planning: reads every configured site, validates the +// requests (throws a user-facing Error on bad input), and returns ONLY the +// sites whose groups/membership actually changed — a no-op save rewrites +// nothing. Requests naming unknown siteIds are silently skipped (the site was +// deleted since the form loaded). +export function planSiteMembershipWrites( + paths: Paths, + slug: string, + requests: MembershipRequest[], +): Site[] { + const bySiteId = new Map(requests.map((r) => [r.siteId, r])); + const changed: Site[] = []; + for (const siteId of listSiteIds(paths)) { + const site = getSite(siteId, paths); + const request = bySiteId.get(siteId); + const existing = site.channels.find((c) => c.slug === slug); + + // Resolve the target group (possibly creating one) for a checked site. + let groups = site.groups; + let groupId: string | undefined; + if (request) { + if (request.newGroupName) { + const id = slugify(request.newGroupName); + if (!isValidGroupId(id)) { + throw new Error( + `Could not derive a valid group id from "${request.newGroupName}" (site "${siteId}")`, + ); + } + // Same slug as an existing group -> reuse it (idempotent resubmits). + if (!groups.some((g) => g.id === id)) { + // Matches SiteForm's addGroup default: new groups start unselected + // and without the inline flag (only the seeded default group renders + // its channels as loose chips) — deliberate, not an omission. + groups = [ + ...groups, + { id, name: request.newGroupName, selectedByDefault: false }, + ]; + } + groupId = id; + } else if (request.groupId) { + // Explicit group must still exist — writeSite would silently drop an + // unknown groupId from the membership, so fail loudly instead. + if (!groups.some((g) => g.id === request.groupId)) { + throw new Error( + `Group "${request.groupId}" no longer exists on site "${siteId}"`, + ); + } + groupId = request.groupId; + } + } + + // Next membership list: preserve every other channel's entry untouched and + // keep this channel's existing extra fields (notably `order`). + let channels: SiteChannelMembership[]; + if (!request) { + channels = existing + ? site.channels.filter((c) => c.slug !== slug) + : site.channels; + } else if (existing) { + channels = site.channels.map((c) => { + if (c.slug !== slug) return c; + if (groupId) return { ...c, groupId }; + const { groupId: _drop, ...rest } = c; + return rest; + }); + } else { + channels = [...site.channels, groupId ? { slug, groupId } : { slug }]; + } + + const groupsChanged = groups !== site.groups; + const membershipChanged = request + ? !existing || (existing.groupId ?? undefined) !== groupId + : !!existing; + if (groupsChanged || membershipChanged) { + changed.push({ ...site, groups, channels }); + } + } + return changed; +} + +export async function applySiteWrites( + sites: Site[], + paths: Paths, +): Promise<void> { + for (const site of sites) { + await writeSite(site, paths); + } +} diff --git a/editor/app/channels/new/page.tsx b/editor/app/channels/new/page.tsx @@ -1,11 +1,22 @@ import type { Metadata } from "next"; import Link from "next/link"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { listSites } from "yt-dlp-transcript-common/lib/site"; +import { sortGroups } from "yt-dlp-transcript-common/lib/channelGroups"; import { ChannelFormClient } from "../components/ChannelFormClient"; +import type { SiteMembershipOption } from "../components/SiteMembershipsSection"; import { createChannelAction } from "../actions"; +export const dynamic = "force-dynamic"; export const metadata: Metadata = { title: "New channel — Channels" }; -export default function NewChannelPage() { +export default async function NewChannelPage() { + const sites: SiteMembershipOption[] = listSites(getPaths()).map((s) => ({ + siteId: s.siteId, + siteTitle: s.siteTitle, + defaultGroupId: s.defaultGroupId, + groups: sortGroups(s.groups).map((g) => ({ id: g.id, name: g.name })), + })); return ( <div className="flex flex-col gap-4"> <div className="flex items-center gap-2 text-sm text-muted-foreground"> @@ -19,6 +30,7 @@ export default function NewChannelPage() { <ChannelFormClient action={createChannelAction} submitLabel="Create channel" + sites={sites} /> </div> ); diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -125,6 +125,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { bool(p.abortOnError), str(p.handlingOverride), spec.bucket, + Boolean(p.forceCookies), ); }, "download-from-playlist": (spec) => { diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -21,6 +21,10 @@ import { type SocialLink, } from "yt-dlp-transcript-common/lib/settings"; import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps"; +import { + DEFAULT_COOKIE_MODE, + isCookieMode, +} from "yt-dlp-transcript-common/lib/cookiePolicy"; import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { sanitizeWorkers, @@ -40,6 +44,10 @@ export async function saveSettingsAction( const cookiesFromBrowser = String( formData.get("cookiesFromBrowser") ?? "", ).trim(); + const cookieModeRaw = String(formData.get("cookieMode") ?? "").trim(); + const cookieMode = isCookieMode(cookieModeRaw) + ? cookieModeRaw + : DEFAULT_COOKIE_MODE; const sleepRaw = String( formData.get("sleepBetweenDownloadsSeconds") ?? "", ).trim(); @@ -213,6 +221,7 @@ export async function saveSettingsAction( transcriptionApps: {}, workers, cookiesFromBrowser, + cookieMode, sleepBetweenDownloadsSeconds: sleepParsed, downloadFormat, minFreeDiskGB: minFreeDiskParsed, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -74,11 +74,40 @@ export function SettingsForm({ initial, apps }: Props) { <WorkersField initial={initial.workers} apps={apps} name="workersJson" /> </fieldset> <Field - label="Cookies from browser (retry only)" + label="Cookies from browser" name="cookiesFromBrowser" defaultValue={initial.cookiesFromBrowser} - hint="Browser spec passed to yt-dlp --cookies-from-browser only when the primary download attempt fails with an auth/age error and the channel doesn't have its own cookies set. e.g. firefox, chrome:Default. Leave blank to disable." + hint="Browser spec passed to yt-dlp --cookies-from-browser (e.g. firefox, chrome:Default) for age-restricted, members-only, and private videos. When it is used is set by the cookie mode below. Leave blank for no cookies. Each channel can override both in its Advanced settings." /> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Cookie mode</span> + <select + name="cookieMode" + defaultValue={initial.cookieMode} + className="rounded border border-border bg-card px-2 py-1 text-sm" + > + <option value="when-required"> + When required — retry auth/age failures with cookies + </option> + <option value="always"> + Always — pass cookies on every yt-dlp invocation + </option> + <option value="defer"> + Defer — never in normal runs; collect into "Needs cookies" + </option> + </select> + <span className="text-xs text-muted-foreground"> + <strong>When required</strong> (default) runs cookie-free and retries + only the attempts that fail with an auth/age error.{" "} + <strong>Always</strong> sends cookies with every invocation + (enumeration, metadata prefetch, availability probes, downloads) — + for channels whose listing itself needs auth.{" "} + <strong>Defer</strong> never uses cookies in normal runs: auth-gated + videos are excluded from batches and gathered into each + channel&apos;s &ldquo;Needs cookies&rdquo; bucket, downloaded on + demand with its &ldquo;Download with cookies&rdquo; button. + </span> + </label> <Field label="Sleep between managed downloads (seconds)" name="sleepBetweenDownloadsSeconds" diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx @@ -28,6 +28,7 @@ type GroupRow = { name: string; description: string; selectedByDefault: boolean; + inline: boolean; }; // A featured related-sites group: a label + the sibling siteIds picked into it, @@ -40,6 +41,7 @@ function toGroupRow(g: ChannelGroup): GroupRow { name: g.name, description: g.description ?? "", selectedByDefault: g.selectedByDefault, + inline: g.inline === true, }; } @@ -81,6 +83,7 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { description: g.description.trim() || undefined, selectedByDefault: g.selectedByDefault, order: idx, + inline: g.inline || undefined, })), ); const channelsJson = JSON.stringify( @@ -119,7 +122,9 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { const addGroup = () => setGroups((prev) => [ ...prev, - { id: "", name: "", description: "", selectedByDefault: false }, + // Manually added groups default off; only the seeded FALLBACK_GROUP + // starts with inline on. + { id: "", name: "", description: "", selectedByDefault: false, inline: false }, ]); const removeGroup = (idx: number) => setGroups((prev) => { @@ -470,6 +475,25 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { /> Selected by default </label> + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + checked={g.inline} + onChange={(e) => + setGroups((prev) => + prev.map((row, i) => + i === idx ? { ...row, inline: e.target.checked } : row, + ), + ) + } + className="accent-brand" + /> + Show channels as individual chips + </label> + <p className="text-xs text-muted-foreground"> + Renders each member channel as its own loose chip in the + viewer&apos;s filter row instead of a collapsible group box. + </p> </div> ); })} diff --git a/editor/app/widget/components/MonitorWidget.tsx b/editor/app/widget/components/MonitorWidget.tsx @@ -11,6 +11,7 @@ import { jobKindLabel } from "../../jobs/jobKindLabels"; import type { WorkersPayload, WorkerView } from "../../workers/components/WorkersView"; import { InlineActionButton } from "../../actionable/components/InlineActionButton"; import type { WidgetActionablePayload } from "../../api/widget/actionable/route"; +import type { WidgetSyncPayload } from "../../api/widget/sync/route"; import { buildWidgetQuery, type WidgetConfig } from "../lib/config"; import { WidgetConfigForm } from "./WidgetConfigForm"; import { WidgetControls } from "./WidgetControls"; @@ -93,6 +94,8 @@ export function MonitorWidget({ // a reload re-parses the same config. const [config, setConfig] = useState<WidgetConfig>(initialConfig); const [settingsOpen, setSettingsOpen] = useState(false); + // Drives the relative "5m ago" labels in the sync readouts; null until mounted. + const now = useNow(); const pollMs = config.pollSeconds * 1000; // The controls row needs live `paused` state, so poll workers whenever either @@ -126,6 +129,16 @@ export function MonitorWidget({ Math.max(pollMs, 15000), null, ); + // Freshness/scheduler markers move on the order of minutes, so poll no faster + // than every 15s regardless of the configured cadence. `refetchSync` lets the + // Sync button refresh the readouts the moment a sweep is queued. + const { data: syncData, refetch: refetchSync } = + usePolledPayload<WidgetSyncPayload>( + "/api/widget/sync", + config.lastSync || config.scheduler, + Math.max(pollMs, 15000), + null, + ); // Apply a config change and mirror it into the address bar so the widget is // self-describing: a reload re-parses the same config and the link stays @@ -167,8 +180,12 @@ export function MonitorWidget({ // A low-disk warning is worth showing even when nothing is running, since it // explains why no downloads start — so it overrides hideIdle. Interactive - // controls also stay visible when idle (you may want to pause preemptively). - if (config.hideIdle && nothingActive && !disk?.low && !config.controls) { + // controls also stay visible when idle (you may want to pause preemptively), + // as do the Sync button and the freshness/scheduler readouts — checking sync + // health is exactly what you do when nothing's active. + const keepWhenIdle = + config.controls || config.sync || config.lastSync || config.scheduler; + if (config.hideIdle && nothingActive && !disk?.low && !keepWhenIdle) { return ( <div className="relative p-2 text-xs text-muted-foreground" @@ -183,12 +200,23 @@ export function MonitorWidget({ return ( <div className="relative flex flex-col gap-3 p-2 text-foreground"> {settingsLayer} - {config.controls && ( + {(config.controls || config.sync) && ( <WidgetControls + controls={config.controls} + sync={config.sync} + channel={config.channel} + confirmSyncAll={config.syncConfirm} paused={workersPayload?.paused ?? false} onWorkersChange={refetchWorkers} + onSynced={refetchSync} /> )} + {config.lastSync && syncData && ( + <LastSyncStrip data={syncData} now={now} absolute={config.syncTimeAbsolute} /> + )} + {config.scheduler && syncData && ( + <SchedulerStrip data={syncData} now={now} absolute={config.syncTimeAbsolute} /> + )} {config.disk && disk?.enabled && <DiskStrip disk={disk} />} {config.cleanable && cleanablePayload && ( <CleanableStrip bytes={cleanablePayload.bytes} /> @@ -329,6 +357,118 @@ function DiskStrip({ disk }: { disk: DiskStatusView }) { ); } +// Compact relative label. `deltaMs = now - target`: positive is past ("5m +// ago"), negative is future ("in 5m"). Steps s → m → h → d. +function relativeTime(deltaMs: number): string { + const past = deltaMs >= 0; + const s = Math.round(Math.abs(deltaMs) / 1000); + const val = + s < 60 + ? `${s}s` + : s < 3600 + ? `${Math.round(s / 60)}m` + : s < 86400 + ? `${Math.round(s / 3600)}h` + : `${Math.round(s / 86400)}d`; + return past ? `${val} ago` : `in ${val}`; +} + +// Format an epoch-ms marker per the widget's abstime flag. Absolute is +// now-independent (locale string); relative depends on the live clock, so it +// returns null until `now` is set (post-mount) to avoid an SSR/hydration +// mismatch and a nonsensical "0s ago" flash. Callers render "…" for null. +function fmtTime( + ms: number, + now: number | null, + absolute: boolean, +): string | null { + if (absolute) return new Date(ms).toLocaleString(); + if (now === null) return null; + return relativeTime(now - ms); +} + +// Last-sync freshness readout: the last full "Sync all" sweep, plus the newest +// individual channel sync when it is more recent than that sweep (a per-channel +// Sync since the last global one). +function LastSyncStrip({ + data, + now, + absolute, +}: { + data: WidgetSyncPayload; + now: number | null; + absolute: boolean; +}) { + const full = + data.lastSyncAllAt === null + ? "never" + : (fmtTime(data.lastSyncAllAt, now, absolute) ?? "…"); + const showChannel = + data.lastIndividualSyncAt !== null && + (data.lastSyncAllAt === null || + data.lastIndividualSyncAt > data.lastSyncAllAt); + return ( + <section + aria-label="Last sync" + className="flex flex-col gap-0.5 rounded border border-border bg-card px-2 py-1 text-xs text-muted-foreground" + > + <span className="tabular-nums">Last full sync: {full}</span> + {showChannel && ( + <span className="tabular-nums"> + Last channel sync:{" "} + {fmtTime(data.lastIndividualSyncAt as number, now, absolute) ?? "…"} + </span> + )} + </section> + ); +} + +// Auto-sync scheduler status: on/off (colored dot), next due, last run. Dims to +// a neutral line when the scheduler is disabled. +function SchedulerStrip({ + data, + now, + absolute, +}: { + data: WidgetSyncPayload; + now: number | null; + absolute: boolean; +}) { + const s = data.scheduler; + const nextText = s.overdue + ? "due now" + : s.nextRunAt !== null + ? (fmtTime(s.nextRunAt, now, absolute) ?? "…") + : "—"; + const lastText = + s.lastRunAt !== null ? (fmtTime(s.lastRunAt, now, absolute) ?? "…") : "never"; + return ( + <section + aria-label="Auto-sync scheduler" + className={`flex items-center gap-2 rounded border px-2 py-1 text-xs ${ + s.enabled + ? "border-border bg-card text-muted-foreground" + : "border-border bg-card text-muted-foreground/60" + }`} + > + <span + className={`inline-block h-2 w-2 shrink-0 rounded-full ${ + s.enabled ? "bg-success" : "bg-muted-foreground/50" + }`} + /> + <span className="tabular-nums"> + Auto-sync {s.enabled ? "on" : "off"} + {s.enabled && ( + <> + {" · "}next {nextText} + {" · "}last run {lastText} + </> + )} + </span> + </section> + ); +} + type CleanablePayload = { bytes: number }; function CleanableStrip({ bytes }: { bytes: number }) { diff --git a/editor/app/widget/components/WidgetConfigForm.tsx b/editor/app/widget/components/WidgetConfigForm.tsx @@ -94,6 +94,35 @@ export function WidgetConfigForm({ /> </fieldset> + <fieldset className="flex flex-col gap-2"> + <legend className="text-sm font-medium mb-1">Sync</legend> + <Check + label="Sync button" + checked={config.sync} + onChange={(v) => onChange({ sync: v })} + /> + <Check + label="Confirm before Sync all" + checked={config.syncConfirm} + onChange={(v) => onChange({ syncConfirm: v })} + /> + <Check + label="Last-sync readout" + checked={config.lastSync} + onChange={(v) => onChange({ lastSync: v })} + /> + <Check + label="Scheduler status" + checked={config.scheduler} + onChange={(v) => onChange({ scheduler: v })} + /> + <Check + label="Absolute timestamps" + checked={config.syncTimeAbsolute} + onChange={(v) => onChange({ syncTimeAbsolute: v })} + /> + </fieldset> + <label className="flex flex-col gap-1 text-sm"> <span className="font-medium">Channel filter (slug, optional)</span> <input diff --git a/editor/app/widget/components/WidgetControls.tsx b/editor/app/widget/components/WidgetControls.tsx @@ -1,30 +1,149 @@ "use client"; +import { useState } from "react"; import { PauseTranscriptionsButton } from "../../jobs/components/PauseTranscriptionsButton"; import { DrainAllButton } from "../../jobs/components/DrainAllButton"; import { RetryAllFailedButton } from "../../jobs/components/RetryAllFailedButton"; +import { syncAllChannelsAction, type SyncAllResult } from "../../channels/actions"; +import { syncAction } from "../../channels/[slug]/pipelineActions"; -// Opt-in interactive controls for the monitor widget (enabled with controls=1). -// Reuses the same actions/buttons as the Workers and Active Jobs pages so a -// pinned widget can free the GPU (pause), wind work down (drain), or recover -// failures (retry) without opening the full app. `onWorkersChange` refetches the -// widget's worker payload so the pause/resume label flips immediately instead of -// waiting for the next poll. +// Opt-in interactive controls for the monitor widget. Two independent +// capabilities, each behind its own flag: +// - `controls` → Pause/Resume + Drain + Retry (reused verbatim from the +// Workers / Active Jobs pages), so a pinned widget can free the GPU, wind +// work down, or recover failures without opening the full app. +// - `sync` → a channel-aware Sync button: pinned to one channel it syncs just +// that channel (draining the per-channel stream); otherwise it sweeps all +// channels. +// `onWorkersChange` refetches the worker payload so the pause/resume label flips +// immediately; `onSynced` refetches the last-sync/scheduler readouts so they +// update as soon as a sweep is queued. export function WidgetControls({ + controls, + sync, + channel, + confirmSyncAll, paused, onWorkersChange, + onSynced, }: { + controls: boolean; + sync: boolean; + channel?: string; + confirmSyncAll: boolean; paused: boolean; onWorkersChange: () => void | Promise<void>; + onSynced: () => void | Promise<void>; }) { return ( <section aria-label="Controls" className="flex flex-wrap items-center gap-1.5" > - <PauseTranscriptionsButton paused={paused} onChange={onWorkersChange} /> - <DrainAllButton /> - <RetryAllFailedButton /> + {controls && ( + <> + <PauseTranscriptionsButton paused={paused} onChange={onWorkersChange} /> + <DrainAllButton /> + <RetryAllFailedButton /> + </> + )} + {sync && ( + <SyncButton + channel={channel} + confirmSyncAll={confirmSyncAll} + onSynced={onSynced} + /> + )} </section> ); } + +// The Sync button. Channel-aware: with `channel` set it runs the streaming +// per-channel sync (draining the stream to completion, exactly as +// ChannelSyncButton); otherwise it runs the full Sync-all sweep (optionally +// gated by a confirm) and briefly surfaces the queued/skipped counts. +function SyncButton({ + channel, + confirmSyncAll, + onSynced, +}: { + channel?: string; + confirmSyncAll: boolean; + onSynced: () => void | Promise<void>; +}) { + const [running, setRunning] = useState(false); + const [error, setError] = useState<string | null>(null); + const [result, setResult] = useState<SyncAllResult | null>(null); + + async function handleClick() { + setError(null); + setResult(null); + if (channel) { + setRunning(true); + try { + const res = await syncAction(channel); + if (!res.ok) { + setError(res.error); + return; + } + // Drain the stream to completion; the job keeps running server-side. + const reader = res.stream.getReader(); + while (true) { + const { done } = await reader.read(); + if (done) break; + } + } catch (e) { + setError((e as Error).message); + } finally { + setRunning(false); + await onSynced(); + } + return; + } + if (confirmSyncAll && !window.confirm("Sync all channels now?")) return; + setRunning(true); + try { + const res = await syncAllChannelsAction(); + setResult(res); + } catch (e) { + setError((e as Error).message); + } finally { + setRunning(false); + await onSynced(); + } + } + + const label = channel ? "Sync" : "Sync all"; + const ariaLabel = channel ? `sync ${channel}` : "sync all channels"; + return ( + <div className="flex items-center gap-2"> + <button + type="button" + onClick={handleClick} + disabled={running} + aria-label={ariaLabel} + className="px-3 py-2 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50" + > + {running ? "Syncing…" : label} + </button> + {result && ( + <span + aria-label="sync all result" + className="text-xs text-muted-foreground tabular-nums" + title={ + result.skipped.length === 0 + ? undefined + : result.skipped.map((s) => `${s.slug}: ${s.reason}`).join("\n") + } + > + Queued {result.queued.length} · skipped {result.skipped.length} + </span> + )} + {error && ( + <span role="alert" className="text-xs text-destructive"> + {error} + </span> + )} + </div> + ); +} diff --git a/editor/app/widget/lib/config.ts b/editor/app/widget/lib/config.ts @@ -42,6 +42,21 @@ export type WidgetConfig = { // Show the in-widget settings gear that opens the config overlay in place. On // by default; turn off to bake a locked-down shared/embedded link. settings: boolean; + // Show the interactive Sync button. Channel-aware: syncs just `channel` when + // the widget is pinned to one, else sweeps all channels. Off by default. + sync: boolean; + // Show the last-sync readout strip (last full sync-all + last individual + // channel sync when newer). Off by default. + lastSync: boolean; + // Show the auto-sync scheduler status strip (on/off, next due, last run). Off + // by default. + scheduler: boolean; + // Readouts show absolute (locale) time instead of relative "5m ago". Off by + // default (relative). + syncTimeAbsolute: boolean; + // window.confirm before a full Sync-all sweep. Off by default. (A pinned + // single-channel Sync is never gated.) + syncConfirm: boolean; }; export const WIDGET_DEFAULTS: WidgetConfig = { @@ -61,6 +76,11 @@ export const WIDGET_DEFAULTS: WidgetConfig = { controls: false, actionable: false, settings: true, + sync: false, + lastSync: false, + scheduler: false, + syncTimeAbsolute: false, + syncConfirm: false, }; // Next's searchParams give each key as string | string[] | undefined. @@ -106,6 +126,11 @@ export function parseWidgetConfig(params: RawParams): WidgetConfig { controls: parseBool(params.controls, WIDGET_DEFAULTS.controls), actionable: parseBool(params.act, WIDGET_DEFAULTS.actionable), settings: parseBool(params.gear, WIDGET_DEFAULTS.settings), + sync: parseBool(params.sync, WIDGET_DEFAULTS.sync), + lastSync: parseBool(params.lastsync, WIDGET_DEFAULTS.lastSync), + scheduler: parseBool(params.sched, WIDGET_DEFAULTS.scheduler), + syncTimeAbsolute: parseBool(params.abstime, WIDGET_DEFAULTS.syncTimeAbsolute), + syncConfirm: parseBool(params.syncask, WIDGET_DEFAULTS.syncConfirm), }; } @@ -141,5 +166,15 @@ export function buildWidgetQuery(config: WidgetConfig): string { sp.set("act", config.actionable ? "1" : "0"); if (config.settings !== WIDGET_DEFAULTS.settings) sp.set("gear", config.settings ? "1" : "0"); + if (config.sync !== WIDGET_DEFAULTS.sync) + sp.set("sync", config.sync ? "1" : "0"); + if (config.lastSync !== WIDGET_DEFAULTS.lastSync) + sp.set("lastsync", config.lastSync ? "1" : "0"); + if (config.scheduler !== WIDGET_DEFAULTS.scheduler) + sp.set("sched", config.scheduler ? "1" : "0"); + if (config.syncTimeAbsolute !== WIDGET_DEFAULTS.syncTimeAbsolute) + sp.set("abstime", config.syncTimeAbsolute ? "1" : "0"); + if (config.syncConfirm !== WIDGET_DEFAULTS.syncConfirm) + sp.set("syncask", config.syncConfirm ? "1" : "0"); return sp.toString(); } diff --git a/editor/e2e/channel-site-membership.spec.ts b/editor/e2e/channel-site-membership.spec.ts @@ -0,0 +1,186 @@ +import { test, expect, type Page } from "@playwright/test"; +import { readJson, resetData, writeSite } from "./helpers"; + +// The channel form's "Sites" membership section: per-site checkbox + group +// dropdown (+ inline "+ New group…") on both the create and edit screens. + +type SiteFile = { + groups: { id: string; name: string; selectedByDefault: boolean }[]; + channels: { slug: string; groupId?: string; order?: number }[]; +}; + +// Sentinel select value for "+ New group…" (SiteMembershipsSection.tsx). +const NEW_GROUP = "__new__"; + +// Two sites over the shared two-channel pool. Alpha has a second group and +// slow-b sits in it with an explicit order; beta has just the default group. +async function seed() { + await resetData("two-slow-channels"); + await writeSite("alpha", { + siteTitle: "Alpha", + groups: [ + { id: "default", name: "All channels", selectedByDefault: true }, + { id: "news", name: "News", selectedByDefault: false }, + ], + channels: [ + { slug: "slow-a" }, + { slug: "slow-b", groupId: "news", order: 5 }, + ], + }); + await writeSite("beta", { siteTitle: "Beta" }); +} + +const readSite = (id: string) => + readJson<SiteFile>(`test-transcripts/sites/${id}/site.json`); + +// The membership inputs are controlled React state that the edit page also +// server-renders, so a pre-hydration click can be silently swallowed (see the +// charts-metadata lost-click incident). Both helpers verify each interaction +// against React-rendered output (the group select / new-group input only +// exist when state says so) and retry until it sticks. +async function setSiteChecked(page: Page, site: string, on: boolean) { + const groupSelect = page.getByLabel(`Group for ${site}`); + await expect(async () => { + if (((await groupSelect.count()) > 0) !== on) { + await page.getByLabel(`Include on ${site}`).click(); + } + await expect(groupSelect).toHaveCount(on ? 1 : 0, { timeout: 1_000 }); + }).toPass({ timeout: 15_000 }); +} + +async function selectGroup(page: Page, site: string, value: string) { + const groupSelect = page.getByLabel(`Group for ${site}`); + const nameInput = page.getByLabel(`New group name for ${site}`); + // Selecting "+ New group…" reveals a React-rendered input — proof the change + // handler ran. Once that sticks, picking the real value is safe. + await expect(async () => { + await groupSelect.selectOption(NEW_GROUP); + await expect(nameInput).toBeVisible({ timeout: 1_000 }); + }).toPass({ timeout: 15_000 }); + if (value !== NEW_GROUP) { + await groupSelect.selectOption(value); + await expect(nameInput).toHaveCount(0); + } +} + +test("create: active site pre-checked, explicit group + second site", async ({ + page, +}) => { + await seed(); + await page.goto("/channels/new?site=alpha"); + // The pre-check happens client-side only (mount effect), so this assertion + // doubles as a hydration gate for the plain interactions below. + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); + await expect(page.getByLabel("Include on Beta")).not.toBeChecked(); + + await page.getByLabel("Group for Alpha").selectOption("news"); + await page.getByLabel("Include on Beta").check(); + await page.getByLabel(/^name/i).fill("Gamma Channel"); + await page.getByLabel(/^slug/i).fill("gamma"); + await page.getByRole("button", { name: /create channel/i }).click(); + await page.waitForURL("**/channels/gamma", { timeout: 10_000 }); + + const alpha = await readSite("alpha"); + expect(alpha.channels).toContainEqual({ slug: "gamma", groupId: "news" }); + const beta = await readSite("beta"); + expect(beta.channels).toContainEqual({ slug: "gamma" }); +}); + +test("create: + New group creates the group and assigns the channel", async ({ + page, +}) => { + await seed(); + await page.goto("/channels/new?site=alpha"); + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); + + await selectGroup(page, "Alpha", NEW_GROUP); + await page.getByLabel("New group name for Alpha").fill("Interviews"); + await page.getByLabel(/^name/i).fill("Delta Channel"); + await page.getByLabel(/^slug/i).fill("delta"); + await page.getByRole("button", { name: /create channel/i }).click(); + await page.waitForURL("**/channels/delta", { timeout: 10_000 }); + + const alpha = await readSite("alpha"); + expect(alpha.groups).toContainEqual({ + id: "interviews", + name: "Interviews", + selectedByDefault: false, + }); + expect(alpha.channels).toContainEqual({ + slug: "delta", + groupId: "interviews", + }); +}); + +test("edit: switching news → (default) keeps order, sibling untouched", async ({ + page, +}) => { + await seed(); + await page.goto("/channels/slow-b"); + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); + await expect(page.getByLabel("Group for Alpha")).toHaveValue("news"); + + await selectGroup(page, "Alpha", ""); + await page.getByRole("button", { name: /save changes/i }).click(); + + await expect + .poll(async () => + (await readSite("alpha")).channels.find((c) => c.slug === "slow-b"), + ) + .toEqual({ slug: "slow-b", order: 5 }); + const alpha = await readSite("alpha"); + expect(alpha.channels).toContainEqual({ slug: "slow-a" }); +}); + +test("edit: unchecking removes the membership, sibling intact", async ({ + page, +}) => { + await seed(); + await page.goto("/channels/slow-b"); + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); + + await setSiteChecked(page, "Alpha", false); + await page.getByRole("button", { name: /save changes/i }).click(); + + await expect + .poll(async () => (await readSite("alpha")).channels.map((c) => c.slug)) + .toEqual(["slow-a"]); +}); + +test("edit: + New group on another site", async ({ page }) => { + await seed(); + await page.goto("/channels/slow-a"); + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); + await expect(page.getByLabel("Include on Beta")).not.toBeChecked(); + + await setSiteChecked(page, "Beta", true); + await selectGroup(page, "Beta", NEW_GROUP); + await page.getByLabel("New group name for Beta").fill("Interviews"); + await page.getByRole("button", { name: /save changes/i }).click(); + + await expect + .poll(async () => (await readSite("beta")).groups) + .toContainEqual({ + id: "interviews", + name: "Interviews", + selectedByDefault: false, + }); + const beta = await readSite("beta"); + expect(beta.channels).toContainEqual({ + slug: "slow-a", + groupId: "interviews", + }); +}); + +test("zero sites: informational note, create still succeeds", async ({ + page, +}) => { + await resetData("empty"); + await page.goto("/channels/new"); + await expect(page.getByText(/no sites configured/i)).toBeVisible(); + + await page.getByLabel(/^name/i).fill("Solo Channel"); + await page.getByLabel(/^slug/i).fill("solo"); + await page.getByRole("button", { name: /create channel/i }).click(); + await page.waitForURL("**/channels/solo", { timeout: 10_000 }); +}); diff --git a/editor/e2e/channels.spec.ts b/editor/e2e/channels.spec.ts @@ -74,12 +74,14 @@ test("requires typed-confirmation to delete", async ({ page }) => { await resetData("one-youtube-channel"); await page.goto("/channels/test-youtube"); - await page.getByPlaceholder("test-youtube").fill("wrong-slug"); + // getByPlaceholder would be ambiguous: the rename form's confirm input + // shares the slug placeholder. + await page.getByLabel("confirm slug to delete").fill("wrong-slug"); await page.getByRole("button", { name: /delete channel/i }).click(); await expect(page.getByText(/type the channel slug/i)).toBeVisible(); expect(await pathExists("test-transcripts/channels/test-youtube")).toBe(true); - await page.getByPlaceholder("test-youtube").fill("test-youtube"); + await page.getByLabel("confirm slug to delete").fill("test-youtube"); await page.getByRole("button", { name: /delete channel/i }).click(); await page.waitForURL("**/channels", { timeout: 10_000 }); expect(await pathExists("test-transcripts/channels/test-youtube")).toBe(false); diff --git a/editor/e2e/cookies-mode.spec.ts b/editor/e2e/cookies-mode.spec.ts @@ -0,0 +1,331 @@ +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import path from "node:path"; +import { test, expect } from "@playwright/test"; +import { + pathExists, + readJson, + resetData, + resolvePath, + writeSettings, +} from "./helpers"; + +// Configurable cookies-from-browser: global + per-channel value, and a cookie +// MODE (always / when-required / defer). The fake yt-dlp's `cookiegated` +// sentinel fails any metadata/download invocation without +// --cookies-from-browser (yt-dlp's age-gate error) and succeeds with it, and +// every relevant invocation line logs `cookies=<value>` so flag presence can +// be asserted per spawn. + +const CHANNEL = "test-cookies"; +const FIXTURE = "cookies-mode-channel"; +const ROOT = `test-transcripts/channels/${CHANNEL}`; +const COOKIES = "firefox:test"; +const here = path.dirname(fileURLToPath(import.meta.url)); + +type DownloadOutcome = { + status: string; + attempts: Array<{ kind: string; usedCookies: boolean }>; +}; + +type Snapshot = { + buckets?: { needsCookies?: string[] }; +}; + +async function defaultSettings(): Promise<Record<string, unknown>> { + const raw = await readFile( + path.join(here, "fixtures", "test-settings.default.json"), + "utf8", + ); + return JSON.parse(raw) as Record<string, unknown>; +} + +async function readInvocations(): Promise<string> { + try { + return await readFile( + path.join(here, "..", ROOT, "fake-ytdlp.invocations"), + "utf8", + ); + } catch { + return ""; + } +} + +// The download job arms a debounced (~1s) snapshot regen; poll snapshot.json +// until the needsCookies bucket reaches the expected size before reloading. +async function waitForNeedsCookiesBucket(size: number): Promise<void> { + await expect + .poll( + async () => { + const snap = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch( + () => null, + ); + return snap?.buckets?.needsCookies?.length ?? -1; + }, + { timeout: 15_000 }, + ) + .toBe(size); +} + +test("settings: cookie value + mode round-trip through the form", async ({ + page, +}) => { + await resetData(FIXTURE); + await page.goto("/settings"); + await page.getByLabel(/^cookies from browser/i).fill(COOKIES); + // The value field's hint mentions "cookie mode", so a label lookup is + // ambiguous — target the select by name. + const modeSelect = page.locator('select[name="cookieMode"]'); + await modeSelect.selectOption("always"); + await page.getByRole("button", { name: /save settings/i }).click(); + await expect( + page.getByRole("status").filter({ hasText: "Saved" }), + ).toBeVisible(); + + const saved = await readJson<{ + cookiesFromBrowser?: string; + cookieMode?: string; + }>("test-settings.json"); + expect(saved.cookiesFromBrowser).toBe(COOKIES); + expect(saved.cookieMode).toBe("always"); + + await page.reload(); + await expect(page.getByLabel(/^cookies from browser/i)).toHaveValue(COOKIES); + await expect(modeSelect).toHaveValue("always"); +}); + +test("channel form: overrides persist and clear back to inherit", async ({ + page, +}) => { + await resetData(FIXTURE); + await page.goto(`/channels/${CHANNEL}`); + await page.locator("summary").filter({ hasText: "Advanced" }).click(); + await page.getByLabel(/^cookies from browser/i).fill("chrome:Profile 1"); + await page.getByLabel("cookie mode").selectOption("defer"); + await page.getByRole("button", { name: /save changes/i }).click(); + + await expect + .poll(async () => + readJson<{ cookiesFromBrowser?: string; cookieMode?: string }>( + `${ROOT}/config.json`, + ), + ) + .toMatchObject({ + cookiesFromBrowser: "chrome:Profile 1", + cookieMode: "defer", + }); + + // Clear both back to inherit. CHANNEL_FORM_FIELDS layering must actually + // remove the stored values, not re-layer the stale ones. + await page.reload(); + await page.locator("summary").filter({ hasText: "Advanced" }).click(); + const cookiesField = page.getByLabel(/^cookies from browser/i); + await expect(cookiesField).toHaveValue("chrome:Profile 1"); + await cookiesField.fill(""); + await page.getByLabel("cookie mode").selectOption(""); + await page.getByRole("button", { name: /save changes/i }).click(); + + await expect + .poll(async () => { + const config = await readJson<Record<string, unknown>>( + `${ROOT}/config.json`, + ); + return { + cookiesFromBrowser: config.cookiesFromBrowser, + cookieMode: config.cookieMode, + }; + }) + .toEqual({ cookiesFromBrowser: undefined, cookieMode: undefined }); +}); + +test("always mode: every yt-dlp invocation carries the cookie value", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData(FIXTURE); + await writeSettings({ + ...(await defaultSettings()), + cookiesFromBrowser: COOKIES, + cookieMode: "always", + }); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + "Managed download complete: 2 succeeded, 0 failed", + { timeout: 60_000 }, + ); + + const invocations = await readInvocations(); + expect(invocations).toMatch( + new RegExp(`prefetch:.*vidcookiegated1 cookies=${COOKIES}`), + ); + expect(invocations).toMatch( + new RegExp(`prefetch:.*normalvideo1 cookies=${COOKIES}`), + ); + expect(invocations).toMatch( + new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`), + ); + expect(invocations).toMatch( + new RegExp(`youtube-single:.*normalvideo1 cookies=${COOKIES}`), + ); + // No cookie-less invocation lines at all in always mode. + expect(invocations).not.toMatch(/cookies=$/m); + + // Prophylactic cookies on a clean success stay "ok" — "ok-with-cookies" + // keeps meaning "cookies were needed". + const outcome = await readJson<DownloadOutcome>( + `${ROOT}/data/vidcookiegated1/download-outcome.json`, + ); + expect(outcome.status).toBe("ok"); + expect(outcome.attempts.every((a) => a.usedCookies)).toBe(true); +}); + +test("when-required: failed prefetch is retried with cookies -> ok-with-cookies", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData(FIXTURE); + await writeSettings({ + ...(await defaultSettings()), + cookiesFromBrowser: COOKIES, + // cookieMode omitted -> back-compat default "when-required" + }); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + `Metadata prefetch auth-required (needs_auth); retrying with --cookies-from-browser ${COOKIES}`, + { timeout: 60_000 }, + ); + await expect(log).toContainText( + "Managed download complete: 2 succeeded, 0 failed", + { timeout: 60_000 }, + ); + + // Two prefetch lines for the gated video: cookie-less first, cookies second. + const invocations = await readInvocations(); + const gatedPrefetches = invocations + .split("\n") + .filter((l) => l.startsWith("prefetch:") && l.includes("vidcookiegated1")); + expect(gatedPrefetches).toHaveLength(2); + expect(gatedPrefetches[0]).toMatch(/cookies=$/); + expect(gatedPrefetches[1]).toContain(`cookies=${COOKIES}`); + // The recovered prefetch hands cookies straight to the real download. + expect(invocations).toMatch( + new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`), + ); + // The normal video never sees cookies in when-required mode. + const normalLines = invocations + .split("\n") + .filter((l) => l.includes("normalvideo1") && l.includes("cookies=")); + expect(normalLines.length).toBeGreaterThan(0); + expect(normalLines.every((l) => l.endsWith("cookies="))).toBe(true); + + const outcome = await readJson<DownloadOutcome>( + `${ROOT}/data/vidcookiegated1/download-outcome.json`, + ); + expect(outcome.status).toBe("ok-with-cookies"); + expect( + outcome.attempts.some( + (a) => a.kind === "metadata-prefetch-auth-retry" && a.usedCookies, + ), + ).toBe(true); +}); + +test("defer: excluded from batches, surfaced in Needs cookies, downloads via the bucket button", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData(FIXTURE); + await writeSettings({ + ...(await defaultSettings()), + cookiesFromBrowser: COOKIES, + cookieMode: "defer", + }); + + // Run 1: the gated video fails cookie-less (no prefetch retry, no auth + // retry — defer never uses cookies in normal runs). + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + "Managed download complete: 1 succeeded, 1 failed", + { timeout: 60_000 }, + ); + let invocations = await readInvocations(); + expect(invocations).not.toContain(`cookies=${COOKIES}`); + expect( + invocations + .split("\n") + .filter((l) => l.startsWith("prefetch:") && l.includes("vidcookiegated1")), + ).toHaveLength(1); + + // The failure lands the video in the snapshot's needsCookies bucket. + await waitForNeedsCookiesBucket(1); + + // Run 2: the recorded needs_auth video is excluded, not re-attempted. + await page.reload(); + await page.getByRole("button", { name: "Download videos" }).click(); + const log2 = page.getByLabel("Download videos output"); + await expect(log2).toContainText("needs_auth deferred=1", { + timeout: 60_000, + }); + await expect(log2).toContainText("Nothing to fetch.", { timeout: 60_000 }); + + // The Needs-cookies card offers the manual cookie run. + const bucket = page.getByLabel("retry needs cookies bucket"); + await expect(bucket).toBeVisible(); + await bucket + .getByRole("button", { name: /^Download with cookies \(1\)$/ }) + .click(); + const retryLog = page.getByLabel("Retry needs cookies output"); + await expect(retryLog).toContainText( + "Managed download complete: 1 succeeded, 0 failed", + { timeout: 60_000 }, + ); + + expect( + await pathExists(`${ROOT}/data/vidcookiegated1/transcript.en.vtt`), + ).toBe(true); + invocations = await readInvocations(); + expect(invocations).toMatch( + new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`), + ); + + // The bucket empties once the snapshot regenerates. + await waitForNeedsCookiesBucket(0); + await page.reload(); + await expect(page.getByLabel("retry needs cookies bucket")).toHaveCount(0); +}); + +test("members_only videos surface in Needs cookies regardless of mode", async ({ + page, +}) => { + await resetData(FIXTURE); + // Default settings: when-required mode. A members-only playlist video with + // no downloaded artifact must still land in the bucket — cookies from a + // subscribed account could recover it even though batches exclude it. + await writeFile( + resolvePath(`${ROOT}/playlist`), + "https://www.youtube.com/watch?v=vidmembers1\n", + ); + const videoDir = resolvePath(`${ROOT}/data/vidmembers1`); + await mkdir(videoDir, { recursive: true }); + await writeFile( + path.join(videoDir, "availability.json"), + JSON.stringify({ + checkedAt: new Date().toISOString(), + availability: "members_only", + webpageUrl: "https://www.youtube.com/watch?v=vidmembers1", + }), + ); + + await page.goto(`/channels/${CHANNEL}`); + const bucket = page.getByLabel("retry needs cookies bucket"); + await expect(bucket).toBeVisible(); + await expect( + bucket.getByRole("button", { name: /^Download with cookies \(1\)$/ }), + ).toBeVisible(); + await expect(page.getByLabel("needs cookies vidmembers1")).toBeVisible(); +}); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -46,6 +46,27 @@ function lastNonFlag() { return undefined; } +// The `cookiegated` sentinel simulates an age-gated video: any metadata/ +// download invocation for it fails with yt-dlp's age-gate error UNLESS +// --cookies-from-browser was passed. (Deliberately distinct from the +// `needsauth` sentinel, which only affects --dump-json probes and still +// downloads fine — retry-bucket.spec depends on that.) +function cookieArg() { + return arg("--cookies-from-browser") ?? ""; +} + +function cookieGateBlocked(url) { + const lower = (url ?? "").toLowerCase(); + return lower.includes("cookiegated") && !cookieArg(); +} + +function failCookieGate(url) { + process.stderr.write( + `ERROR: [youtube] ${url}: Sign in to confirm your age. This video may be inappropriate for some users.\n`, + ); + process.exit(1); +} + function urlIdYouTube(u) { try { const parsed = new URL(u); @@ -214,7 +235,11 @@ async function modeDownloadFromFile(file, opts = {}) { async function modeDownloadOneUrl(url, opts) { process.stdout.write(`[fake-ytdlp] single-url download ${url}\n`); - await appendFile("fake-ytdlp.invocations", `download-one:${url}\n`); + await appendFile( + "fake-ytdlp.invocations", + `download-one:${url} cookies=${cookieArg()}\n`, + ); + if (cookieGateBlocked(url)) failCookieGate(url); // Record the resolved -f selector so format-selection tests can assert it // (e.g. Odysee "auto" -> original/bestaudio/worst). await appendFile("fake-ytdlp.invocations", `format:${arg("-f") ?? ""}\n`); @@ -409,8 +434,9 @@ async function modeYoutubeSingleUrlManaged(url) { await ensureDir(videoDir); await appendFile( "fake-ytdlp.invocations", - `youtube-single:${url}\n`, + `youtube-single:${url} cookies=${cookieArg()}\n`, ); + if (cookieGateBlocked(url)) failCookieGate(url); process.stdout.write(`[fake-ytdlp] managed single-URL ${id}\n`); // URL sentinels for no-subs-fallback tests. The sentinels live in the @@ -475,7 +501,11 @@ async function main() { if (has("--dump-json") && has("--skip-download")) { const url = lastNonFlag() ?? ""; const lower = url.toLowerCase(); - await appendFile("fake-ytdlp.invocations", `dump-json:${url}\n`); + await appendFile( + "fake-ytdlp.invocations", + `dump-json:${url} cookies=${cookieArg()}\n`, + ); + if (cookieGateBlocked(url)) failCookieGate(url); if (lower.includes("deleted")) { process.stderr.write(`ERROR: [youtube] ${url}: Video unavailable\n`); process.exit(1); @@ -587,7 +617,11 @@ async function main() { const id = urlIdYouTube(url); const videoDir = path.join("data", id); await ensureDir(videoDir); - await appendFile("fake-ytdlp.invocations", `prefetch:${url}\n`); + await appendFile( + "fake-ytdlp.invocations", + `prefetch:${url} cookies=${cookieArg()}\n`, + ); + if (cookieGateBlocked(url)) failCookieGate(url); await writeMetadata(videoDir, id, urlSentinels(url)); process.stdout.write(`[fake-ytdlp] metadata prefetch ${id}\n`); return; diff --git a/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/config.json b/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/config.json @@ -0,0 +1,5 @@ +{ + "handling": "youtube", + "name": "Test Cookies", + "url": "https://www.youtube.com/@example/videos" +} diff --git a/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/playlist b/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/playlist @@ -0,0 +1,2 @@ +https://www.youtube.com/watch?v=vidcookiegated1 +https://www.youtube.com/watch?v=normalvideo1 diff --git a/editor/e2e/site-scope.spec.ts b/editor/e2e/site-scope.spec.ts @@ -76,8 +76,9 @@ test("deploy targets the active site and is disabled under all sites", async ({ await expect( page.getByText(/select a specific site from the sidebar to build & deploy\./i), ).toBeVisible(); + // exact: the "Build & deploy all sites" batch button also matches otherwise. await expect( - page.getByRole("button", { name: "Build & deploy" }), + page.getByRole("button", { name: "Build & deploy", exact: true }), ).toBeDisabled(); await page.goto("/deploy?site=alpha"); @@ -104,6 +105,9 @@ test("creating a channel under a site adds it to that site's membership", async }) => { await twoSites(); await page.goto("/channels/new?site=alpha"); + // The Sites section pre-checks the active site — that's what carries the + // membership now (the old hidden activeSite field is gone). + await expect(page.getByLabel("Include on Alpha")).toBeChecked(); await page.getByLabel(/^name/i).fill("Gamma Channel"); await page.getByLabel(/^slug/i).fill("gamma"); await page.getByLabel(/^url/i).fill("https://www.youtube.com/@gamma/videos"); diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts @@ -241,6 +241,55 @@ test("edit page loads an existing public URL and featured site", async ({ ).toBeChecked(); }); +test("inline-chips flag round-trips; seeded default group has it on", async ({ + page, +}) => { + await resetData("empty"); + + await page.goto("/sites/new"); + await page.getByLabel(/site id/i).fill("chipsite"); + await page.getByLabel(/site title/i).fill("Chip Site"); + await page.getByLabel(/header title/i).fill("Chip Site"); + + const chipToggles = page.getByRole("checkbox", { + name: /show channels as individual chips/i, + }); + // The seeded default group (FALLBACK_GROUP) ships with the flag on. + await expect(chipToggles).toHaveCount(1); + await expect(chipToggles.first()).toBeChecked(); + + // Manually added groups default off. + await page.getByRole("button", { name: /\+ add group/i }).click(); + await expect(chipToggles).toHaveCount(2); + await expect(chipToggles.nth(1)).not.toBeChecked(); + await page.getByLabel("ID", { exact: true }).nth(1).fill("extra"); + await page.getByLabel("Name", { exact: true }).nth(1).fill("Extra"); + + await page.getByRole("button", { name: /create site/i }).click(); + await expect( + page.getByRole("status").filter({ hasText: "Saved" }), + ).toBeVisible(); + + type GroupsSiteFile = { + groups: { id: string; inline?: boolean }[]; + }; + // Poll: the "Saved" status and the on-disk write can land slightly apart. + await expect(async () => { + const site = await readJson<GroupsSiteFile>( + "test-transcripts/sites/chipsite/site.json", + ); + expect(site.groups[0].inline).toBe(true); + // Flag-off groups serialize without the key at all. + expect("inline" in site.groups[1]).toBe(false); + }).toPass({ timeout: 10_000 }); + + // Reload the edit page — checkbox states restore from disk. + await page.goto("/sites/chipsite"); + await expect(chipToggles).toHaveCount(2); + await expect(chipToggles.first()).toBeChecked(); + await expect(chipToggles.nth(1)).not.toBeChecked(); +}); + test("first-run migrate button appears only when no sites exist", async ({ page, }) => { diff --git a/editor/e2e/widget.spec.ts b/editor/e2e/widget.spec.ts @@ -99,6 +99,42 @@ test("controls=1 adds interactive Pause/Resume, Drain, and Retry buttons", async ).toBeVisible({ timeout: 10_000 }); }); +test("sync=1 shows a channel-aware Sync button", async ({ page }) => { + // No channel pinned → sweeps all channels. + await page.goto("/widget?sync=1"); + const syncAll = page.getByRole("button", { name: "sync all channels" }); + await expect(syncAll).toBeVisible(); + await expect(syncAll).toHaveText("Sync all"); + + // Empty corpus → the sweep queues nothing and reports its counts. + await syncAll.click(); + await expect(page.getByLabel("sync all result")).toContainText( + "Queued 0 · skipped 0", + { timeout: 10_000 }, + ); + + // Pinned to one channel → the button targets just that channel. + await page.goto("/widget?sync=1&channel=test-youtube"); + const syncOne = page.getByRole("button", { name: "sync test-youtube" }); + await expect(syncOne).toBeVisible(); + await expect(syncOne).toHaveText("Sync"); +}); + +test("lastsync=1 shows the last-sync readout", async ({ page }) => { + await page.goto("/widget?lastsync=1"); + const strip = page.getByRole("region", { name: "Last sync" }); + await expect(strip).toBeVisible({ timeout: 10_000 }); + // Empty corpus with no sweep recorded → never. + await expect(strip).toContainText("Last full sync: never"); +}); + +test("sched=1 shows the auto-sync scheduler strip", async ({ page }) => { + await page.goto("/widget?sched=1"); + const strip = page.getByRole("region", { name: "Auto-sync scheduler" }); + await expect(strip).toBeVisible({ timeout: 10_000 }); + await expect(strip).toContainText(/Auto-sync (on|off)/); +}); + test("section params gate what renders", async ({ page }) => { await page.goto("/widget?jobs=0"); await expect(page.getByRole("heading", { name: "Workers" })).toBeVisible(); @@ -210,4 +246,21 @@ test("display toggles serialize into the link and preview", async ({ page }) => .check(); await expect(url).toHaveValue(/[?&]controls=1(&|$)/); await expect(preview).toHaveAttribute("src", /[?&]controls=1(&|$)/); + + // The Sync fieldset flags each serialize under their own key. + await page.getByRole("checkbox", { name: "Sync button" }).check(); + await expect(url).toHaveValue(/[?&]sync=1(&|$)/); + await expect(preview).toHaveAttribute("src", /[?&]sync=1(&|$)/); + + await page.getByRole("checkbox", { name: "Confirm before Sync all" }).check(); + await expect(url).toHaveValue(/[?&]syncask=1(&|$)/); + + await page.getByRole("checkbox", { name: "Last-sync readout" }).check(); + await expect(url).toHaveValue(/[?&]lastsync=1(&|$)/); + + await page.getByRole("checkbox", { name: "Scheduler status" }).check(); + await expect(url).toHaveValue(/[?&]sched=1(&|$)/); + + await page.getByRole("checkbox", { name: "Absolute timestamps" }).check(); + await expect(url).toHaveValue(/[?&]abstime=1(&|$)/); }); diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -2,6 +2,12 @@ ## [Unreleased] - **"Ask AI" — the Report panel's citations are now clickable, precise to the exact moment cited.** The chat's answer bubbles already linked their `[1]`, `[2]`… markers to the cited transcript; the **Report** panel — the persistent document Report mode and whole-corpus sweeps maintain — rendered as plain text, so the citations a user works from most were dead. Now the report mirrors the chat: it renders each citation as an in-page link and lists its **numbered sources** beneath the document, and clicking one **opens the transcript modal seeked to the cited line**. Citations are also upgraded — in the report *and* the chat — to a precise-moment **`[n @ mm:ss]`** form (the source's number plus the specific line's timestamp, e.g. `[3 @ 12:34]`), so a citation jumps to the exact moment instead of the video's first matched snippet. A slightly-off model timestamp is **snapped to the nearest real transcript line**, and it stays graceful: a bare `[n]` still resolves to the first snippet, and an out-of-range number or unparseable time stays plain text rather than becoming a broken link. Making this work in the report (a single accumulated string, unlike a chat message that carries its own sources) required two new pieces: a **persisted report source registry** — every video the report cites, deduped and stored light (no excerpt text) — and **global citation numbering** so a report's `[n]` is stable across every batch/turn folded in, not renumbered per write. The registry is saved with the conversation and each saved chat, so a report's clickable citations survive a reload; New chat clears it; and on the federated hub /ask (no player) citations fall back to scroll-to-source, exactly like the chat. See `export/app/ask/citations.tsx` (new shared renderer — `linkifyCitations` + `CitationLink` + `CitationSources`, with `[n @ mm:ss]` parsing and nearest-snippet snapping), `export/app/ask/{MessageBubble,ReportPanel,AskChat,useAskChat,askChatStorage}.tsx/ts`, `export/app/lib/{askRetrieval,askConversation,searchAgent}.ts` (optional/global-registry numbering + `mergeReportSources` + the `[n @ mm:ss]` prompt updates), and `export/e2e/ask-chat.spec.ts`. +- **Loose channel chips for flagged groups, plus uniform chip layout.** A channel group can now be marked `inline` (site.json → manifest), which renders each of its member channels as its own loose chip in the filter row — a checkbox + name styled like a collapsed group chip, with no expand affordance — flowing in sort order among the normal group chips; inline groups are also skipped by the multi-group default-collapse heuristic, so their channels are always visible. The out-of-the-box **"All channels" bucket now behaves this way**: a site with no configured groups switches from one open full-width card to loose per-channel chips (behavior change; add an explicit group to keep the old card). The chips themselves also get a layout fix: instead of shrink-to-fit chips that could line-break their own text, all collapsed chips share a **uniform min width** (`sm:min-w-48`), never wrap internally — long names ellipsize with the full name as a tooltip — counts right-align inside the chip, and below the `sm` breakpoint chips stack one per line at full width while larger screens pack them into rows. See `common/lib/channelGroups.ts` (`ChannelGroup.inline`), `common/components/{WorkspaceSearchBar,SearchSessionContext}.tsx`, and `export/e2e/inline-channel-chips.spec.ts`. + +## [0.8.1] - 2026-07-22 +- **Channel-group filters are now compact chips.** On a site with many small channel groups the filter panel used to stack every group as its own full-width card, so ten groups meant ten rows before you saw a single channel. Collapsed groups now shrink to labeled chips — a chevron, a tri-state checkbox (checked / mixed / unchecked), the group's accent dot + name, and a `selected/total` count — that wrap several per line. The chip's checkbox toggles the **whole group** on or off in one click *without* expanding it (a mixed group's click selects all), and clicking the name expands the group into the familiar full-width card with per-channel checkboxes and All/None. Groups start collapsed on multi-group sites (your own expand/collapse choices still stick); a collapsed chip shows the group's description as a tooltip. Selection stays flat per-channel — profiles, share links, and saved filters are unchanged. See `common/components/{WorkspaceSearchBar,SearchSessionContext}.tsx`, `common/components/ui/checkbox.tsx`, and `export/e2e/channel-group-chips.spec.ts`. + +## [0.8.0] - 2026-07-20 - **MCP server: switch which corpus you're reading on the fly — and it sticks across reconnects.** The MCP server was pinned to a single corpus at launch (`--hub` / `--remote` / `--local` or the `TRANSCRIPT_*` env), so re-aiming it at a different site or a local build meant editing the MCP config and restarting. Now, connected to (say) the **archilyzer hub**, you can retarget it from inside a session with three new **read-only** tools: **`list_sources`** shows the active corpus and, in a hub context, its member sites (`siteId · title · url`); **`use_source`** switches the active corpus — to a single hub member (`site:` — which becomes a plain single-site source and so regains **full channel-group + alias** support), a **federated subset** of the hub (`sites:[…]`, unknown tokens reported not dropped), or an arbitrary `remote:` URL / `local:` dir / `hub:` URL; and **`reset_source`** returns to the startup source. The selection is **persisted** to a small state file (under `$XDG_STATE_HOME/yt-dlp-transcript-mcp`, overridable with `TRANSCRIPT_MCP_STATE_DIR`) keyed by the **startup** source, so it survives a reconnect and two differently-configured servers keep independent selections. Everything the sweep/search tools do reads the current active source. Still strictly **read-only** — this only changes *which* already-published static shards are read, the same capability the startup flags already grant this locally-run tool; nothing is written to any corpus. See `mcp/src/{sources,sourceController,source,server,index}.ts`, `mcp/src/sourceController.test.ts`, and `mcp/README.md`. - **MCP server: run the corpus sweep through Claude Code itself — on plan usage, no API key.** The in-browser "corpus sweep" (fold every matching transcript into a running report) needs a BYO AI key, and a free key conks out fast on a real 30k-video corpus. The `mcp/` server now lets **Claude Code be the sweep engine** instead, via a first-class **`sweep` prompt** (a slash command, `/mcp__<name>__sweep query="k cups" channel="chrissie-mayr"`) plus the tools to drive it. `search_transcripts` gains **paging** — it reports the full `total` and `has_more`, so the whole match set can be enumerated with `offset` (and `include_snippets:false` for a cheap worklist) — and is now **alias-aware**: a plain query that matches a curated search alias also searches the alias regex (e.g. `k cups` → also `cake cup`), with the footer naming which aliases fired; reaching the scan cap is surfaced as **PARTIAL** coverage rather than hidden. A new **`get_transcripts`** tool batch-reads up to 20 videos in one call as bounded, timestamped **excerpt windows** around the matches (alias-correct) — or full transcripts without a query — so a sweep stays token-bounded. The `sweep` prompt walks Claude through search → enumerate → plan `ceil(N/batch)` batches → per-batch windowed read + cross-referenced upsert of cited findings (*title + [mm:ss]*) into a markdown report it maintains with its own Write/Edit tools. The sweep is now **group-aware and multi-channel**: `search_transcripts` scope is additive over one-or-more channels (`channel`/`channels`) and/or channel **groups** (`group`/`groups`, matched by group **id or display name** — `group="other"` ≡ `group="Extended Universe"` — and expanded to the group's channels), with the footer naming the resolved scope and flagging any channel/group token that matched nothing so a typo isn't silently a whole-corpus scan; `get_transcripts` takes matching `channels` lookup hints; and `list_channels` now organizes channels under their groups with a compact `id · name · N channels` cheat-sheet. Crucially the `sweep` prompt is now **pick-first**: invoked with **no** scope it makes Claude list the groups/channels and **ask which to sweep — or to confirm the whole corpus — before enumerating**, rather than silently scanning everything (an explicit `search_transcripts` call with no selector still means "all"). (Hub mode defers per-site group resolution — multi-channel scoping still works there.) The MCP stays strictly **read-only**; only the report file is written, in Claude's own working directory. See `mcp/src/{search,source,server}.ts`, `mcp/src/search.test.ts`, and `mcp/README.md`. - **"Ask AI" now paces itself to your key's rate limit instead of failing.** A whole-corpus sweep on a free-tier key used to fire provider calls as fast as the loop could produce them, blow straight through the per-minute request cap, and stop dead on the first HTTP 429 (and a rate-limit *mid-answer* surfaced as a hard error, because only the sweep path ever handled 429). Now a single client-side limiter sits behind **every** AI call: it **spaces requests** to a conservative, free-tier-safe **requests-per-minute** default chosen per model (e.g. Gemini `*-pro` → 5/min, `*flash` → 10/min; Claude/OpenAI higher), so a paid key runs fast and a free key just runs *slowly* rather than erroring. When a 429 does land, the limiter **honours the provider's own retry hint** (the `Retry-After` header, or Gemini's `RetryInfo` retry-delay) — or an exponential backoff — and **re-issues the request** (safe for streaming: the retry happens before any answer text is emitted), and it **self-lowers** the rate after a 429 so later calls pace slower. Only a 429 that outlasts the retries falls through to the existing **pause/checkpoint** (the right home for a daily-quota cap you resume tomorrow). Provider settings gain an editable **Requests per minute** field (with the effective spacing, e.g. *"≈12s between requests"*, and a reset-to-default), and a paced sweep shows a **"Rate limited — retrying in Ns"** line in its progress strip so it never looks frozen. See `export/app/lib/rateLimit.ts` (the whole mechanism), `export/app/lib/askProvider.ts` (`PausableError` + `acquire`/retry in `askStream`, 429 in `ensureOk`), `export/app/lib/nativeTools/shared.ts` (`acquire`/retry in `postJson`), `export/app/ask/{useAskChat,ProviderSettings,PinnedResultsPanel,AskChat}.tsx`, and `export/e2e/ask-workspace.spec.ts`. diff --git a/export/e2e/channel-group-chips.spec.ts b/export/e2e/channel-group-chips.spec.ts @@ -0,0 +1,256 @@ +import { expect, test, type Page, type Route } from "@playwright/test"; + +// Self-contained fixtures exercising the channel-group filter chips +// (common/components/WorkspaceSearchBar.tsx): several small groups collapse to +// wrapping chips with a tri-state group checkbox; expanding breaks a group out +// to a full-width card with per-channel checkboxes and All/None. + +type Chan = { name: string; slug: string; groupId: string }; + +const CHANNELS: Chan[] = [ + { name: "Alpha", slug: "alpha", groupId: "g1" }, + { name: "Beta", slug: "beta", groupId: "g1" }, + { name: "Gamma", slug: "gamma", groupId: "g2" }, + { name: "Delta", slug: "delta", groupId: "g3" }, +]; + +const GROUPS = [ + { + id: "g1", + name: "Group One", + description: "First fixture group", + selectedByDefault: true, + order: 1, + }, + { + id: "g2", + name: "Group Two", + description: "Second fixture group", + selectedByDefault: true, + order: 2, + }, + { + id: "g3", + name: "Group Three", + description: "Third fixture group", + selectedByDefault: true, + order: 3, + }, +]; + +function summary(c: Chan) { + return { + slug: `${c.slug}/vid-${c.slug}`, + id: `vid-${c.slug}`, + channelSlug: c.slug, + title: `Video from ${c.name}`, + uploadDate: "20250101", + date: "2025-01-01", + duration: "5:00", + channel: c.name, + isLivestream: false, + ageRestricted: false, + isDeleted: false, + isUnlisted: false, + platform: "youtube" as const, + webpageUrl: `https://example.com/vid-${c.slug}`, + }; +} + +async function fulfillJson(route: Route, body: unknown) { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify(body), + }); +} + +async function installRoutes(page: Page) { + const list = CHANNELS.map(summary); + await page.route("**/summaries/manifest.json", async (route) => { + await fulfillJson(route, { + version: 3, + totalCount: list.length, + pageSize: 1000, + pageCount: 1, + generatedAt: new Date().toISOString(), + channels: CHANNELS.map((c) => ({ + name: c.name, + count: 1, + groupId: c.groupId, + })), + groups: GROUPS, + defaultGroupId: "g1", + }); + }); + await page.route("**/summaries/page-*.json", async (route) => { + await fulfillJson(route, list); + }); + // Transcripts are never fetched by these tests, but stub the routes so + // nothing falls through to the dev server. + await page.route(/\/transcripts\/[^/]+\/manifest\.json$/, async (route) => { + await fulfillJson(route, { + version: 1, + channelSlug: "unused", + pageCount: 0, + maxPageBytes: 8388608, + generatedAt: new Date().toISOString(), + slugToPage: {}, + }); + }); + await page.route(/\/transcripts\/[^/]+\/page-\d+\.json$/, async (route) => { + await fulfillJson(route, []); + }); + // No live chat. + await page.route("**/subs/manifest.json", async (route) => { + await fulfillJson(route, { + version: 4, + channels: [], + totalCount: 0, + liveChatTotalCount: 0, + generatedAt: new Date().toISOString(), + }); + }); +} + +async function waitForHydration(page: Page) { + await page.getByTestId("query-builder").waitFor(); +} + +// Group cards are <details> nested inside the outer Filters <details>. +function groupCard(page: Page, name: string) { + return page.locator("details details").filter({ hasText: name }); +} + +function groupCheckbox(page: Page, name: string) { + return page.getByRole("checkbox", { name: `Select all in ${name}` }); +} + +// A member-channel checkbox inside its group card. Scoped + exact so the +// per-result "Select "Video from Alpha" for AI" checkboxes never collide. +function memberCheckbox(page: Page, group: string, channel: string) { + return groupCard(page, group).getByRole("checkbox", { + name: channel, + exact: true, + }); +} + +// Expand/collapse by clicking the group's name inside the chip summary — +// the tri-state checkbox and All/None isolate their own clicks. +async function clickGroupName(page: Page, name: string) { + await groupCard(page, name) + .locator("summary") + .getByText(name, { exact: true }) + .click(); +} + +test.describe("channel-group filter chips", () => { + test.beforeEach(async ({ page }) => { + await installRoutes(page); + await page.goto("/"); + await waitForHydration(page); + }); + + test("groups start collapsed as chips that wrap onto one row", async ({ + page, + }) => { + // All three chips render with their selected counts. + await expect(groupCard(page, "Group One")).toContainText("2/2"); + await expect(groupCard(page, "Group Two")).toContainText("1/1"); + await expect(groupCard(page, "Group Three")).toContainText("1/1"); + + // Collapsed by default on a multi-group site: member checkboxes hidden. + await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden(); + await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden(); + + // Uniform min-width chips pack onto a shared row. + const one = await groupCard(page, "Group One") + .locator("summary") + .boundingBox(); + const two = await groupCard(page, "Group Two") + .locator("summary") + .boundingBox(); + expect(one).not.toBeNull(); + expect(two).not.toBeNull(); + expect(one!.y).toBe(two!.y); + }); + + test("chip checkbox toggles the whole group without expanding", async ({ + page, + }) => { + const box = groupCheckbox(page, "Group One"); + await expect(box).toHaveAttribute("aria-checked", "true"); + + // Fully selected → click deselects the whole group. + await box.click(); + await expect(groupCard(page, "Group One")).toContainText("0/2"); + await expect(box).toHaveAttribute("aria-checked", "false"); + // ...and the chip did not expand. + await expect(groupCard(page, "Group One")).not.toHaveAttribute("open"); + await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden(); + + // Unselected → click selects the whole group. + await box.click(); + await expect(groupCard(page, "Group One")).toContainText("2/2"); + await expect(box).toHaveAttribute("aria-checked", "true"); + await expect(groupCard(page, "Group One")).not.toHaveAttribute("open"); + }); + + test("partial selection shows indeterminate; clicking selects all", async ({ + page, + }) => { + // Expand Group One and deselect one member. + await clickGroupName(page, "Group One"); + await memberCheckbox(page, "Group One", "Alpha").click(); + await expect(groupCard(page, "Group One")).toContainText("1/2"); + + // Collapse — the chip checkbox reads mixed. + await clickGroupName(page, "Group One"); + await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden(); + const box = groupCheckbox(page, "Group One"); + await expect(box).toHaveAttribute("aria-checked", "mixed"); + + // Radix cycles indeterminate → checked, i.e. "select all". + await box.click(); + await expect(groupCard(page, "Group One")).toContainText("2/2"); + await expect(box).toHaveAttribute("aria-checked", "true"); + }); + + test("expanding shows member checkboxes and All/None at full width", async ({ + page, + }) => { + await clickGroupName(page, "Group One"); + + const card = groupCard(page, "Group One"); + await expect(card).toHaveAttribute("open", ""); + await expect(memberCheckbox(page, "Group One", "Alpha")).toBeVisible(); + await expect(memberCheckbox(page, "Group One", "Beta")).toBeVisible(); + await expect(card.getByRole("button", { name: "All" })).toBeVisible(); + await expect(card.getByRole("button", { name: "None" })).toBeVisible(); + + // The expanded card breaks out to (roughly) the container's full width. + const containerBox = await card.locator("xpath=..").boundingBox(); + const cardBox = await card.boundingBox(); + expect(containerBox).not.toBeNull(); + expect(cardBox).not.toBeNull(); + expect(cardBox!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9); + + // Other groups stay collapsed chips. + await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden(); + }); + + test("expanded/collapsed state persists across a reload", async ({ + page, + }) => { + await clickGroupName(page, "Group Two"); + await expect(groupCard(page, "Group Two")).toHaveAttribute("open", ""); + + await page.reload(); + await waitForHydration(page); + + await expect(groupCard(page, "Group Two")).toHaveAttribute("open", ""); + await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeVisible(); + await expect(groupCard(page, "Group One")).not.toHaveAttribute("open"); + await expect(groupCard(page, "Group Three")).not.toHaveAttribute("open"); + }); +}); diff --git a/export/e2e/inline-channel-chips.spec.ts b/export/e2e/inline-channel-chips.spec.ts @@ -0,0 +1,240 @@ +import { expect, test, type Page, type Route } from "@playwright/test"; + +// Self-contained fixtures exercising inline channel groups +// (common/components/WorkspaceSearchBar.tsx): a group flagged `inline: true` +// renders its member channels as loose individual chips — checkbox + name, +// no collapsible box — flowing among the normal group chips. + +type Chan = { name: string; slug: string; groupId: string }; + +const LONG_NAME = + "Extremely Long Channel Name That Certainly Overflows A Uniform Chip"; + +const CHANNELS: Chan[] = [ + { name: "Epsilon", slug: "epsilon", groupId: "solo" }, + { name: LONG_NAME, slug: "longname", groupId: "solo" }, + { name: "Alpha", slug: "alpha", groupId: "g1" }, + { name: "Beta", slug: "beta", groupId: "g1" }, + { name: "Gamma", slug: "gamma", groupId: "g2" }, +]; + +const GROUPS = [ + { + id: "solo", + name: "Solo", + selectedByDefault: true, + order: 0, + inline: true, + }, + { + id: "g1", + name: "Group One", + description: "First fixture group", + selectedByDefault: true, + order: 1, + }, + { + id: "g2", + name: "Group Two", + description: "Second fixture group", + selectedByDefault: true, + order: 2, + }, +]; + +function summary(c: Chan) { + return { + slug: `${c.slug}/vid-${c.slug}`, + id: `vid-${c.slug}`, + channelSlug: c.slug, + title: `Video from ${c.name}`, + uploadDate: "20250101", + date: "2025-01-01", + duration: "5:00", + channel: c.name, + isLivestream: false, + ageRestricted: false, + isDeleted: false, + isUnlisted: false, + platform: "youtube" as const, + webpageUrl: `https://example.com/vid-${c.slug}`, + }; +} + +async function fulfillJson(route: Route, body: unknown) { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify(body), + }); +} + +async function installRoutes(page: Page) { + const list = CHANNELS.map(summary); + await page.route("**/summaries/manifest.json", async (route) => { + await fulfillJson(route, { + version: 3, + totalCount: list.length, + pageSize: 1000, + pageCount: 1, + generatedAt: new Date().toISOString(), + channels: CHANNELS.map((c) => ({ + name: c.name, + count: 1, + groupId: c.groupId, + })), + groups: GROUPS, + defaultGroupId: "solo", + }); + }); + await page.route("**/summaries/page-*.json", async (route) => { + await fulfillJson(route, list); + }); + // Transcripts are never fetched by these tests, but stub the routes so + // nothing falls through to the dev server. + await page.route(/\/transcripts\/[^/]+\/manifest\.json$/, async (route) => { + await fulfillJson(route, { + version: 1, + channelSlug: "unused", + pageCount: 0, + maxPageBytes: 8388608, + generatedAt: new Date().toISOString(), + slugToPage: {}, + }); + }); + await page.route(/\/transcripts\/[^/]+\/page-\d+\.json$/, async (route) => { + await fulfillJson(route, []); + }); + // No live chat. + await page.route("**/subs/manifest.json", async (route) => { + await fulfillJson(route, { + version: 4, + channels: [], + totalCount: 0, + liveChatTotalCount: 0, + generatedAt: new Date().toISOString(), + }); + }); +} + +async function waitForHydration(page: Page) { + await page.getByTestId("query-builder").waitFor(); +} + +// Group cards are <details> nested inside the outer Filters <details>. +function groupCard(page: Page, name: string) { + return page.locator("details details").filter({ hasText: name }); +} + +// A member-channel checkbox inside its group card. +function memberCheckbox(page: Page, group: string, channel: string) { + return groupCard(page, group).getByRole("checkbox", { + name: channel, + exact: true, + }); +} + +// A loose per-channel chip from an inline group. +function inlineChip(page: Page, channel: string) { + return page + .getByTestId("inline-channel-chip") + .filter({ hasText: channel }) + .first(); +} + +test.describe("inline channel chips", () => { + test.beforeEach(async ({ page }) => { + await installRoutes(page); + await page.goto("/"); + await waitForHydration(page); + }); + + test("inline group renders loose chips instead of a group chip", async ({ + page, + }) => { + // Two loose chips, one per Solo member — no <details> card for "Solo" + // and no group bulk-toggle checkbox. + await expect(page.getByTestId("inline-channel-chip")).toHaveCount(2); + await expect(groupCard(page, "Solo")).toHaveCount(0); + await expect( + page.getByRole("checkbox", { name: "Select all in Solo" }), + ).toHaveCount(0); + + // Inline members are visible on first load: the default-collapse + // heuristic only counts collapsible groups (still >1 here, so g1/g2 + // members stay hidden behind collapsed chips). + await expect( + inlineChip(page, "Epsilon").getByRole("checkbox"), + ).toBeVisible(); + await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden(); + await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden(); + }); + + test("inline chip toggles its channel without expanding anything", async ({ + page, + }) => { + await expect(page.getByText("5 of 5 channels")).toBeVisible(); + + const box = inlineChip(page, "Epsilon").getByRole("checkbox"); + await expect(box).toHaveAttribute("aria-checked", "true"); + + await box.click(); + await expect(page.getByText("4 of 5 channels")).toBeVisible(); + await expect(box).toHaveAttribute("aria-checked", "false"); + // No chip expanded as a side effect of the toggle. + await expect(page.locator("details details[open]")).toHaveCount(0); + + await box.click(); + await expect(page.getByText("5 of 5 channels")).toBeVisible(); + await expect(box).toHaveAttribute("aria-checked", "true"); + await expect(page.locator("details details[open]")).toHaveCount(0); + }); + + test("uniform chip widths with long names ellipsized", async ({ page }) => { + // Short inline chip and short collapsed group chip share the uniform + // min width. + const epsilon = await inlineChip(page, "Epsilon").boundingBox(); + const g2 = await groupCard(page, "Group Two").boundingBox(); + expect(epsilon).not.toBeNull(); + expect(g2).not.toBeNull(); + expect(epsilon!.width).toBe(g2!.width); + + // Long-named chip carries the full name as a tooltip, never wraps the + // text (same height as a short chip), and stays inside the container. + const long = inlineChip(page, LONG_NAME); + await expect(long).toHaveAttribute("title", LONG_NAME); + const longBox = await long.boundingBox(); + const containerBox = await long.locator("xpath=..").boundingBox(); + expect(longBox).not.toBeNull(); + expect(containerBox).not.toBeNull(); + expect(longBox!.height).toBe(epsilon!.height); + expect(longBox!.width).toBeLessThanOrEqual(containerBox!.width); + }); +}); + +test.describe("inline channel chips on mobile", () => { + test.use({ viewport: { width: 390, height: 844 } }); + + test.beforeEach(async ({ page }) => { + await installRoutes(page); + await page.goto("/"); + await waitForHydration(page); + }); + + test("chips stack one per line at phone widths", async ({ page }) => { + const epsilon = await inlineChip(page, "Epsilon").boundingBox(); + const long = await inlineChip(page, LONG_NAME).boundingBox(); + const containerBox = await inlineChip(page, "Epsilon") + .locator("xpath=..") + .boundingBox(); + expect(epsilon).not.toBeNull(); + expect(long).not.toBeNull(); + expect(containerBox).not.toBeNull(); + + // Stacked: same left edge, different rows, each near full width. + expect(epsilon!.x).toBe(long!.x); + expect(epsilon!.y).not.toBe(long!.y); + expect(epsilon!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9); + expect(long!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9); + }); +}); diff --git a/mcp/README.md b/mcp/README.md @@ -13,14 +13,35 @@ reads the site's already-published static JSON shards (`corpus.json` + | Tool | What it does | |------|--------------| | `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. | -| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). | -| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. `channel`/`channels` hints speed the lookup. | -| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). | +| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). | +| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`, or compact `link_style:"base"` stamps; `max_lines` caps the excerpt), or full transcripts without a query. `channel`/`channels` hints speed the lookup. | +| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). | | `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. | +| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity). **Previews** a plan by default; **applies** it (switch source + search, linked results) on `apply:true`. Adjust in natural language via `overrides`. | | `list_sources` | Show the **active** corpus (label + kind + target) and, with a hub context, its member sites (`siteId · title · url`, marking the current subset) so you can pick one to switch to. | | `use_source` | **Switch which corpus is read**, on the fly — a hub member (`site`), a hub subset (`sites`), or an arbitrary `remote` URL / `local` dir / `hub` URL. Persists across reconnects. | | `reset_source` | Return to the source the server was started with and clear the persisted selection. | +**Clickable moment links.** Every timestamp the read tools emit is a Markdown link +to the exact second — an **archilyzer viewer** deep link (`…/?v=<slug>&t=<sec>`, +opening the transcript modal at the moment) when the source has a public origin, +otherwise the video's platform watch page with a per-platform time param +(YouTube `&t=<sec>s`, Odysee/Twitch equivalents). A `--local` source with no +origin falls back to platform links. + +**Compact base links (`link_style:"base"`).** Inline links are ~70–90 chars *per +line* — bulk an agent pipeline shouldn't pay for. `search_transcripts` and +`get_transcripts` accept `link_style:"base"`: each video gets **one** +`- moment_base:` header line (a URL ending in `t=`) and every stamp becomes the +compact `[mm:ss|<seconds>]`. The expansion rule: **full moment link = +`<moment_base><seconds>`** — append the integer after the `|`, e.g. +`[title @ 2:36](<moment_base>156)`. The seconds are floored exactly like the +inline links', so both styles cite the identical second. A video whose base +can't be built (no viewer origin and a platform whose time param doesn't take +raw seconds — Twitch — or doesn't exist — Rumble/Kick) omits the line: cite its +`- source:` URL plain instead. `open_link` results and `get_transcript` stay +inline-linked (future work). + ### `search_transcripts` Beyond `query`, `regex`, and `limit`: @@ -42,8 +63,11 @@ Beyond `query`, `regex`, and `limit`: full `total` and `has_more`, so you can enumerate a query's *entire* match set: page with `offset += limit` until `has_more` is `no`. - **`include_snippets`** (default true) — set `false` for a cheap worklist - (id / title / channel / date / match count, no cue text). Ideal for the - planning pass of a sweep. + (id / title / channel / date / match count, no cue text — the per-video + `- source:` line is dropped too). Ideal for the planning pass of a sweep. +- **`link_style`** (default `"inline"`) — `"base"` switches to the compact + agent-pipeline form: a `- moment_base:` line per hit and `[mm:ss|<seconds>]` + snippet stamps (see *Compact base links* above). - **`use_aliases`** (default true) — expand the query through the site's curated search aliases. A plain query that matches a curated trigger also searches the alias's regex, so mis-transcribed spellings are caught (e.g. `k cups` also @@ -66,6 +90,46 @@ video's full transcript comes back as markdown. Missing ids are reported inline. lookup when a batch spans several channels (the ids are already scoped by the search that produced them). +- **`link_style`** (default `"inline"`) — `"base"` emits one `- moment_base:` + line per video and compact `[mm:ss|<seconds>]` stamps instead of a full + Markdown link per line (see *Compact base links* above). +- **`max_lines`** (default 200) — cap on merged excerpt lines per video when a + query is given (the earliest lines are kept). Ignored without a query. + +### `open_link` — paste a viewer share link, at full fidelity + +The archilyzer viewer's **Share** button produces a URL that encodes the whole +search: the origin, a `qt=` composite **query tree** (any mix of scopes — +transcripts, live-chat, title/channel, description, tags — combined with +AND/OR/NOT), and the filter block (`fc` channels, `ft` type, `fa` audience, +`fav` availability, `fdf`/`fdt` upload-date range). `open_link` re-runs that exact +search here — no manual source-switching or query reconstruction, nothing lost. + +It is a **preview → adjust → apply** flow: + +- **Preview** (default, `apply:false`) — decode the link and return a *plan*: the + resolved source (origin, hub vs single-site, auto-probed from `corpus.json`), + the query tree rendered readably, every active filter, the channel scope + validated against the live corpus, and anything ignored or warned (the `fk` + subtitle-track token is vestigial in the composite share model — decoded and + reported as ignored). **No source switch, no search yet.** +- **Adjust** — re-call with structured `overrides` to honor a natural-language + edit: `clear_availability` (drop the availability filter), `clear_type`, + `clear_age`, `clear_dates`, `clear_filters`, `clear_channels`, + `channels:[…]` (re-scope), `date_from`/`date_to`, or `query`/`regex`/ + `query_scope` (replace the search). The plan updates in place. +- **Apply** (`apply:true`) — switch the active source to the link's origin (a + single-site remote, or a hub when the origin federates), run the search at full + fidelity (the query tree + filters, honoring the chat and availability scopes), + and return the first page of **linked** results. The applied plan is echoed at + the top for transparency. Pageable via `limit`/`offset`. Still read-only. + +``` +open_link link="https://rekietalyzer.pages.dev/?qt=…&fv=1&fc=Rekieta%20Law&ft=v&fav=a" +open_link link="…" overrides={ "clear_availability": true } # "drop the availability filter" +open_link link="…" apply=true # switch source + search +``` + ## The `sweep` prompt A first-class slash command that turns **Claude Code itself** into the corpus @@ -74,9 +138,12 @@ browser does with a BYO AI key, but driven by your Claude **plan usage** (no API key) and with the report written to a file. Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query` -(required), `channel?`, `channels?` (comma-separated slugs/names), `group?` (a -channel group id or name), `directive?` (default *"key claims & -contradictions"*), `batch_size?` (default 8), `report_path?` (default +(required *unless* a `link` is given), `link?` (an archilyzer viewer share URL to +seed the sweep from), `channel?`, `channels?` (comma-separated slugs/names), +`group?` (a channel group id or name), `directive?` (default *"key claims & +contradictions"*), `batch_size?` (default 8), `parse_model?` (default +`haiku` — the model requested for the per-batch extractor subagents; they only +quote verbatim, so the cheapest model wins), `report_path?` (default `./sweep-report.md`). ``` @@ -84,8 +151,15 @@ contradictions"*), `batch_size?` (default 8), `report_path?` (default /mcp__rekietalyzer__sweep query="k cups" group="other" # a whole group /mcp__rekietalyzer__sweep query="k cups" group="Extended Universe" # …by name /mcp__rekietalyzer__sweep query="k cups" # pick-first (see below) +/mcp__rekietalyzer__sweep link="https://…/?qt=…&fv=1&fc=…" # seed from a share link ``` +**Seed from a share link (`link=`).** Pass a viewer share URL instead of a +`query` and the sweep starts by driving `open_link`: it previews the decoded plan +(source, query tree, filters, scope, ignored bits), lets you confirm or adjust in +natural language, then applies it — switching source and enumerating the full +tree+filter match set — before batching. Everything the link encodes is honored. + **Pick-first when no scope is given.** Because a whole-corpus sweep can be a lot of work, invoking `sweep` **without** a `channel`/`channels`/`group` makes Claude FIRST call `list_channels`, present the groups and their channels, and ask you @@ -97,13 +171,32 @@ name (`group="other"` ≡ `group="Extended Universe"`). Once scoped, the prompt instructs Claude Code to: **search** (aliases auto-expand; the footer names the resolved scope) → **enumerate** the full worklist by paging with `include_snippets:false` until `has_more` is false → -**plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts` for -windowed context, cross-reference against the report so far, and upsert findings -(claims, contradictions, sources cited as *title + [mm:ss]*) into well-titled -`##` sections of the report file with its own Write/Edit tools → finish with a -short summary. The MCP stays read-only; only the report file is written, in +**plan** `ceil(N / batch_size)` batches → **per batch, map-reduce** → finish with +a short summary. The MCP stays read-only; only the report file is written, in Claude's working directory. +**Dumb extractors on the cheap model.** Each batch is handed to a **subagent** +(Claude Code's Task tool) spawned as a *verbatim extractor* on the cheapest +model — the prompt requests `parse_model` (default `haiku`) via the Task tool's +model override, and gracefully spawns on the default when no override exists. +The extractor calls `get_transcripts` with `link_style:"base"` and returns +*only* the directive-relevant lines, **verbatim** — grouped per video under its +`moment_base:` line, in their compact `[mm:ss|sec]` form, under a hard budget of +**≤40 lines (~600 words) per batch**. No analysis, no summarizing: the heavy +transcript bulk lives and dies inside the cheap subagent, and nothing irrelevant +is ever reprocessed. The **orchestrator does all the synthesis** — claims, +contradictions, cross-referencing — expanding each kept citation to a full link +by appending the seconds to the video's `moment_base`, then discards the +fragment. Batches are independent, so several can run in parallel. If no +subagent tool is available, the batch is processed inline (still +`link_style:"base"`) and the raw excerpt text dropped after folding. + +**Linked citations.** Every source is cited as a clickable +`[title @ mm:ss](<moment url>)` link — built by appending the cited integer +seconds to the video's `moment_base` from the tool output, so a click seeks to +the exact second (see *Compact base links* above; a video with no base is cited +by its `source:` URL). Note `open_link` results stay inline-linked. + ## Data source (pick one) Resolved from flags or env — precedence hub > remote > local: diff --git a/mcp/src/momentUrl.test.ts b/mcp/src/momentUrl.test.ts @@ -0,0 +1,219 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + momentUrl, + viewerMomentUrl, + platformMomentUrl, + momentBaseUrl, + viewerMomentBaseUrl, + platformMomentBaseUrl, +} from "yt-dlp-transcript-common/lib/momentUrl"; + +// ─── Archilyzer viewer link (preferred) ─── + +test("viewer link: origin + slug + seconds → ?v=<slug>&t=<sec>", () => { + const url = viewerMomentUrl( + "https://rekietalyzer.pages.dev", + "rekietalaw/0eYLR6CbR78", + 754, + ); + assert.ok(url); + const u = new URL(url!); + assert.equal(u.origin, "https://rekietalyzer.pages.dev"); + assert.equal(u.pathname, "/"); + // The slug round-trips through URL decoding to exactly the modal's ?v= value. + assert.equal(u.searchParams.get("v"), "rekietalaw/0eYLR6CbR78"); + assert.equal(u.searchParams.get("t"), "754"); +}); + +test("viewer link: an origin with its own path/query is normalized to /", () => { + const url = viewerMomentUrl("https://site.example/search?q=x", "ch/vid", 10); + const u = new URL(url!); + assert.equal(u.pathname, "/"); + assert.equal(u.searchParams.get("q"), null); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("viewer link: t is omitted at second 0", () => { + const url = viewerMomentUrl("https://site.example", "ch/vid", 0); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), null); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("viewer link: null without an origin or a slug", () => { + assert.equal(viewerMomentUrl(null, "ch/vid", 10), null); + assert.equal(viewerMomentUrl("https://site.example", null, 10), null); +}); + +// ─── Platform fallback ─── + +test("platform link: YouTube gets &t=<sec>s appended", () => { + const url = platformMomentUrl( + "https://www.youtube.com/watch?v=abc123", + "youtube", + 90, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("v"), "abc123"); + assert.equal(u.searchParams.get("t"), "90s"); +}); + +test("platform link: platform is detected from the URL when not given", () => { + const url = platformMomentUrl("https://www.youtube.com/watch?v=abc123", null, 5); + assert.match(url!, /t=5s/); +}); + +test("platform link: Odysee gets ?t=<sec> (raw seconds)", () => { + const url = platformMomentUrl( + "https://odysee.com/@chan/video", + "odysee", + 42, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), "42"); +}); + +test("platform link: Twitch gets ?t=<XhYmZs>", () => { + const url = platformMomentUrl( + "https://www.twitch.tv/videos/12345", + "twitch", + 3661, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), "1h1m1s"); +}); + +test("platform link: Rumble has no reliable start param → bare webpage url", () => { + const url = platformMomentUrl("https://rumble.com/v123-title.html", "rumble", 60); + assert.equal(url, "https://rumble.com/v123-title.html"); +}); + +test("platform link: second 0 → bare webpage url; no webpage url → null", () => { + assert.equal( + platformMomentUrl("https://www.youtube.com/watch?v=abc", "youtube", 0), + "https://www.youtube.com/watch?v=abc", + ); + assert.equal(platformMomentUrl(null, "youtube", 30), null); +}); + +// ─── momentUrl: prefers the viewer link, falls back to the platform link ─── + +test("momentUrl: viewer link wins when an origin is present", () => { + const url = momentUrl({ + siteOrigin: "https://site.example", + slug: "ch/vid", + seconds: 12, + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + const u = new URL(url!); + assert.equal(u.origin, "https://site.example"); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("momentUrl: falls back to the platform link with no origin (local source)", () => { + const url = momentUrl({ + siteOrigin: null, + slug: "ch/vid", + seconds: 12, + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + assert.match(url!, /youtube\.com/); + assert.match(url!, /t=12s/); +}); + +test("momentUrl: null when neither a viewer nor a platform link can be built", () => { + const url = momentUrl({ siteOrigin: null, slug: "ch/vid", seconds: 12 }); + assert.equal(url, null); +}); + +// ─── Base (appendable) forms — the link_style:"base" building blocks ─── + +test("viewer base: ends in t= with the slug encoded; appending seconds parses", () => { + const base = viewerMomentBaseUrl( + "https://rekietalyzer.pages.dev", + "rekietalaw/uamorPe6hSc", + ); + assert.ok(base); + assert.ok(base!.endsWith("&t="), "the base ends in t= so seconds append cleanly"); + assert.ok( + base!.includes("v=rekietalaw%2FuamorPe6hSc"), + "the slug's / is %2F-encoded by URLSearchParams", + ); + // Appending integer seconds yields exactly the viewer deep-link params. + const u = new URL(base! + "754"); + assert.equal(u.searchParams.get("v"), "rekietalaw/uamorPe6hSc"); + assert.equal(u.searchParams.get("t"), "754"); +}); + +test("viewer base: null without an origin/slug or with an unparseable origin", () => { + assert.equal(viewerMomentBaseUrl(null, "ch/vid"), null); + assert.equal(viewerMomentBaseUrl("https://site.example", null), null); + assert.equal(viewerMomentBaseUrl("not a url", "ch/vid"), null); +}); + +test("platform base: YouTube ends in t= (existing query → &t=, none → ?t=)", () => { + const withQuery = platformMomentBaseUrl( + "https://www.youtube.com/watch?v=abc123", + "youtube", + ); + assert.ok(withQuery!.endsWith("&t=")); + const noQuery = platformMomentBaseUrl("https://youtu.be/abc123", "youtube"); + assert.ok(noQuery!.endsWith("?t=")); +}); + +test("platform base: a pre-existing t param is stripped and re-appended LAST", () => { + const base = platformMomentBaseUrl( + "https://www.youtube.com/watch?t=99s&v=abc", + "youtube", + ); + assert.ok(base!.endsWith("&t="), "t= is the last param"); + assert.ok(!base!.includes("99s"), "the old seek value is gone"); + const u = new URL(base! + "42"); + assert.equal(u.searchParams.get("v"), "abc"); + assert.equal(u.searchParams.get("t"), "42"); +}); + +test("platform base: Odysee ends in t= (raw-seconds param)", () => { + const base = platformMomentBaseUrl("https://odysee.com/@chan/video", "odysee"); + assert.ok(base!.endsWith("?t=")); +}); + +test("platform base: non-appendable cases → null (never an invalid seek)", () => { + // Twitch's t takes an XhYmZs token — appending raw seconds would mis-seek. + assert.equal( + platformMomentBaseUrl("https://www.twitch.tv/videos/12345", "twitch"), + null, + ); + // Rumble has no dependable start param at all. + assert.equal( + platformMomentBaseUrl("https://rumble.com/v123-title.html", "rumble"), + null, + ); + assert.equal(platformMomentBaseUrl(null, "youtube"), null); + assert.equal(platformMomentBaseUrl("not a url", "youtube"), null); +}); + +test("momentBaseUrl: viewer base wins, platform fallback, else null", () => { + const viewer = momentBaseUrl({ + siteOrigin: "https://site.example", + slug: "ch/vid", + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + assert.ok(viewer!.startsWith("https://site.example/?v=")); + assert.ok(viewer!.endsWith("&t=")); + + const platform = momentBaseUrl({ + siteOrigin: null, + slug: "ch/vid", + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + assert.ok(platform!.startsWith("https://www.youtube.com/watch?v=vid")); + assert.ok(platform!.endsWith("&t=")); + + assert.equal(momentBaseUrl({ siteOrigin: null, slug: "ch/vid" }), null); +}); diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts @@ -2,29 +2,60 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; -import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; +import type { + ChannelTranscriptsManifest, + ChannelSubsManifest, +} from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups"; -import type { ChannelGroups, ChannelRef, ShardSource } from "./source"; +import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; +import type { + ChannelGroups, + ChannelRef, + ShardSource, + VideoAvailability, +} from "./source"; import { searchTranscripts, getWindowedTranscript, buildMatcher, resolveSelectedChannels, + runSearchSpec, + type SearchFilters, } from "./search"; import { createServer } from "./server"; +// Every filter facet kept (the "no-op" filter) — override fields per test. +const KEEP_ALL: SearchFilters = { + videos: true, + livestreams: true, + allAges: true, + restricted: true, + available: true, + unlisted: true, + deleted: true, +}; + // ─── A tiny in-memory ShardSource for the tests ─── function cues(...pairs: [number, string][]): Cue[] { return pairs.map(([start, text]) => ({ start, end: start + 3, text })); } -function vid(id: string, title: string, channelSlug: string, cs: Cue[]): TranscriptDetail { +function vid( + id: string, + title: string, + channelSlug: string, + cs: Cue[], + extra: Partial<TranscriptDetail> = {}, +): TranscriptDetail { return { - slug: id, + // Real records carry a channel-prefixed slug (`<channelSlug>/<id>`); mirror + // that so viewer-link + availability joins are exercised. + slug: `${channelSlug}/${id}`, id, channelSlug, title, @@ -38,6 +69,7 @@ function vid(id: string, title: string, channelSlug: string, cs: Cue[]): Transcr platform: "youtube", webpageUrl: `https://example.test/${id}`, cues: cs, + ...extra, }; } @@ -51,16 +83,40 @@ const K_CUPS_ALIAS: SearchAlias = { }; // Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup". +// Extra metadata on a1/a3/a4/a5 exercises the description/tags scopes and the +// type/age/date filters without changing what matches the "coffee" cue search. const CHAN_A: TranscriptDetail[] = [ - vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"])), + vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), { + description: "a pour over brewing guide", + tags: ["espresso", "beans"], + }), vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])), - vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"])), - vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"])), - vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"])), + vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), { + isLivestream: true, + }), + vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), { + ageRestricted: true, + }), + vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"]), { + uploadDate: "20250601", + }), vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])), vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])), ]; +// Live-chat cues, keyed by video slug — served via the subs shard methods so a +// `chat`-scope leaf has something to match. Only b1 has a chat track. +const CHAT: Record<string, Cue[]> = { + "chan-b/b1": cues([12, "@fan: banana bread anyone"], [15, "@mod: stay on topic"]), +}; + +// Availability overrides, keyed by video slug (defaults: available). a1 deleted, +// a2 unlisted — the rest available. +const AVAILABILITY: Record<string, VideoAvailability> = { + "chan-a/a1": { isDeleted: true, isUnlisted: false }, + "chan-a/a2": { isDeleted: false, isUnlisted: true }, +}; + // Channel B: one long transcript for windowing (a single match at 100s). const CHAN_B: TranscriptDetail[] = [ vid( @@ -136,6 +192,52 @@ class StubSource implements ShardSource { async transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { return this.pages(ch)[page] ?? []; } + + publicOrigin(): string | null { + return null; // a stub has no viewer origin + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + // Only chan-b ships a (single-page) subs shard, with b1 on page 0. + if (ch.slug !== "chan-b") return null; + return { + version: 1, + channelSlug: ch.slug, + pageCount: 1, + maxPageBytes: 0, + generatedAt: "", + tracks: ["live_chat"], + slugToPage: { b1: 0 }, + }; + } + + async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + if (ch.slug !== "chan-b" || page !== 0) return []; + const slug = "chan-b/b1"; + return [ + { + slug, + id: "b1", + channelSlug: "chan-b", + title: "Long one", + uploadDate: "20240101", + date: "2024-01-01", + duration: "5:00", + channel: "Channel B", + isLivestream: false, + ageRestricted: false, + isDeleted: false, + isUnlisted: false, + platform: "youtube", + webpageUrl: "https://example.test/b1", + tracks: { live_chat: CHAT[slug] }, + }, + ]; + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + return new Map(Object.entries(AVAILABILITY)); + } } // ─── (a) Paging: offset / limit / total / hasMore ─── @@ -302,6 +404,202 @@ test("getWindowedTranscript: no matches yields no lines", async () => { assert.equal(lines.length, 0); }); +test("getWindowedTranscript: a stamp formatter renders the bracket contents", async () => { + const rec = CHAN_B[0]; + const { match } = buildMatcher({ query: "coffee" }); + const { lines } = getWindowedTranscript(rec, match, { + before: 30, + after: 30, + stamp: (clock, seconds) => `${clock}|${Math.floor(seconds)}`, + }); + assert.ok( + lines.includes("[1:40|100] here is the coffee moment"), + lines.join("\n"), + ); +}); + +test("getWindowedTranscript: maxLines keeps the earliest merged lines", async () => { + const rec = CHAN_B[0]; + const { match } = buildMatcher({ query: "coffee" }); + const { lines, matchCount } = getWindowedTranscript(rec, match, { + before: 30, + after: 30, + maxLines: 2, + }); + assert.equal(matchCount, 1, "matchCount is unaffected by the line cap"); + assert.equal(lines.length, 2); + assert.ok(lines[0].includes("right before the moment")); + assert.ok(lines[1].includes("here is the coffee moment")); + assert.ok(!lines.join("\n").includes("right after the moment")); +}); + +// ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ─── + +async function allChannels(src: StubSource): Promise<ChannelRef[]> { + return (await resolveSelectedChannels(src, {})).channels; +} + +function leafTree( + query: string, + scope: "transcripts" | "chat" | "metadata" | "description" | "tags", + extra: Record<string, unknown> = {}, +) { + return newGroup({ children: [newLeaf({ query, scope, ...extra })] }); +} + +async function specIds( + src: StubSource, + tree: ReturnType<typeof newGroup>, + spec: { filters?: SearchFilters | null; aliases?: SearchAlias[] } = {}, +): Promise<string[]> { + const res = await runSearchSpec(src, await allChannels(src), { tree, ...spec }, { limit: 50 }); + return res.hits.map((h) => h.videoId).sort(); +} + +test("spec transcripts leaf: matches cue text (same set as the plain scanner)", async () => { + const src = new StubSource(); + assert.deepEqual(await specIds(src, leafTree("coffee", "transcripts")), [ + "a1", "a2", "a3", "a4", "a5", "b1", + ]); +}); + +test("spec metadata leaf: matches title/channel, not cues", async () => { + const src = new StubSource(); + // "kcup" is in a6's title but no cue (cue says "k cups" with a space). + assert.deepEqual(await specIds(src, leafTree("kcup", "metadata")), ["a6"]); +}); + +test("spec description leaf: matches the description only", async () => { + const src = new StubSource(); + // "brewing" is only in a1's description, in no cue. + assert.deepEqual(await specIds(src, leafTree("brewing", "description")), ["a1"]); +}); + +test("spec tags leaf: matches the joined tags", async () => { + const src = new StubSource(); + assert.deepEqual(await specIds(src, leafTree("espresso", "tags")), ["a1"]); +}); + +test("spec chat leaf: matches live_chat cues fetched from subs shards", async () => { + const src = new StubSource(); + // "banana" only appears in b1's live chat, in no transcript cue. + const res = await runSearchSpec(src, await allChannels(src), { + tree: leafTree("banana", "chat"), + }, { limit: 50 }); + assert.deepEqual(res.hits.map((h) => h.videoId), ["b1"]); + assert.equal(res.hits[0].snippets[0].scope, "chat"); + assert.equal(res.hits[0].snippets[0].track, "live_chat"); +}); + +test("spec AND: transcripts AND tags narrows to the intersection", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags" }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a1"]); +}); + +test("spec OR: description OR chat unions the two", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "OR", + children: [ + newLeaf({ query: "brewing", scope: "description" }), + newLeaf({ query: "banana", scope: "chat" }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a1", "b1"]); +}); + +test("spec negate: coffee AND NOT (espresso tag) drops a1", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags", negate: true }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a2", "a3", "a4", "a5", "b1"]); +}); + +test("spec filter ft: livestreams:false drops the livestream (a3)", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, livestreams: false }, + }); + assert.deepEqual(ids, ["a1", "a2", "a4", "a5", "b1"]); +}); + +test("spec filter fa: keep only age-restricted → a4", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, allAges: false }, + }); + assert.deepEqual(ids, ["a4"]); +}); + +test("spec filter fav: available-only drops deleted a1 + unlisted a2", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, unlisted: false, deleted: false }, + }); + assert.deepEqual(ids, ["a3", "a4", "a5", "b1"]); +}); + +test("spec filter fav: deleted+unlisted-only keeps a1 + a2", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, available: false }, + }); + assert.deepEqual(ids, ["a1", "a2"]); +}); + +test("spec filter dates: fdf bound keeps only the later upload (a5)", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, dateFrom: "20250101" }, + }); + assert.deepEqual(ids, ["a5"]); +}); + +test("spec paging/total: stable total with an offset/limit slice", async () => { + const src = new StubSource(); + const chans = await allChannels(src); + const tree = leafTree("coffee", "transcripts"); + const page = await runSearchSpec(src, chans, { tree }, { limit: 2, offset: 2 }); + assert.equal(page.total, 6); + assert.deepEqual(page.hits.map((h) => h.videoId), ["a3", "a4"]); + assert.equal(page.hasMore, true); +}); + +test("spec aliases: a transcripts leaf expands via curated aliases", async () => { + const src = new StubSource(); + const res = await runSearchSpec( + src, + await allChannels(src), + { tree: leafTree("k cups", "transcripts"), aliases: await src.loadAliases() }, + { limit: 50 }, + ); + assert.deepEqual(res.hits.map((h) => h.videoId).sort(), ["a6", "a7"]); + assert.equal(res.firedAliases.length, 1); + assert.equal(res.firedAliases[0].id, "k-cups"); +}); + +test("spec hits carry slug + platform for moment links", async () => { + const src = new StubSource(); + const res = await runSearchSpec(src, await allChannels(src), { + tree: leafTree("coffee", "transcripts"), + }, { limit: 1 }); + assert.equal(res.hits[0].slug, "chan-a/a1"); + assert.equal(res.hits[0].platform, "youtube"); + assert.ok(res.hits[0].snippets[0].seconds >= 0); +}); + // ─── End-to-end through the MCP server (tools + prompts + missing ids) ─── async function connectClient(source: ShardSource): Promise<Client> { @@ -410,6 +708,8 @@ test("server: the sweep prompt lists with its arguments and renders the query", "channels", "directive", "group", + "link", + "parse_model", "query", "report_path", ]); @@ -449,3 +749,159 @@ test("server: the sweep prompt with no scope instructs a group/channel pick firs assert.match(text, /confirm \*\*all\*\*/); await client.close(); }); + +test("server: the sweep prompt mandates linked citations and subagent batches", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { query: "k cups", channel: "chan-a" }, + }); + const text = (got.messages[0].content as { text: string }).text; + // Feature 1: linked citation form (not the old "title + [mm:ss]"). + assert.match(text, /\[title @ mm:ss\]\(<moment url>\)/); + assert.ok(!/title \+ \[mm:ss\]/.test(text), "old bare citation form is gone"); + // Feature 2: subagent map-reduce + inline fallback. + assert.match(text, /Spawn a subagent/); + assert.match(text, /Task tool/); + assert.match(text, /Fallback/); + await client.close(); +}); + +test("server: the sweep prompt accepts a link= seed and drives open_link", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { link: "https://site.example/?q=coffee" }, + }); + const text = (got.messages[0].content as { text: string }).text; + assert.match(text, /open_link/); + assert.match(text, /apply:true/); + assert.match(text, /https:\/\/site\.example\/\?q=coffee/); + // A link-only sweep needs no query argument. + assert.match(text, /share link/); + await client.close(); +}); + +// ─── link_style:"base" — compact stamps, moment_base, max_lines, worklist trim ─── + +test("server: get_transcripts link_style base emits moment_base + compact stamps", async () => { + const client = await connectClient(new StubSource()); + const res = await client.callTool({ + name: "get_transcripts", + arguments: { video_ids: ["b1"], query: "coffee", link_style: "base" }, + }); + const out = firstText(res); + // StubSource has no viewer origin → the platform (YouTube-param) base. + assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m); + assert.match(out, /\[1:40\|100\] here is the coffee moment/); + assert.ok(!out.includes("](http"), "no full inline links in base excerpts"); + assert.match(out, /<moment_base><seconds>/, "the expansion note is present"); + await client.close(); +}); + +test("server: get_transcripts default output stays inline-linked (regression pin)", async () => { + const client = await connectClient(new StubSource()); + const res = await client.callTool({ + name: "get_transcripts", + arguments: { video_ids: ["b1"], query: "coffee" }, + }); + const out = firstText(res); + assert.ok(out.includes("](https://example.test/b1?t=100s)"), out); + assert.ok(!out.includes("moment_base"), "no base artifacts in inline mode"); + await client.close(); +}); + +test("server: get_transcripts base full transcript uses compact stamps + header base", async () => { + const client = await connectClient(new StubSource()); + const res = await client.callTool({ + name: "get_transcripts", + arguments: { video_ids: ["b1"], link_style: "base" }, + }); + const out = firstText(res); + assert.match(out, /\[0:00\|0\] intro chatter/); + assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m); + assert.ok(!out.includes("](http"), "no inline links in base full transcripts"); + await client.close(); +}); + +test("server: get_transcripts max_lines caps the merged excerpt lines", async () => { + const client = await connectClient(new StubSource()); + const res = await client.callTool({ + name: "get_transcripts", + arguments: { video_ids: ["b1"], query: "coffee", max_lines: 1 }, + }); + const out = firstText(res); + assert.match(out, /1 matching line\(s\), windowed/); + assert.ok(out.includes("right before the moment"), "the earliest line is kept"); + assert.ok(!out.includes("right after the moment"), "later lines are dropped"); + await client.close(); +}); + +test("server: search_transcripts link_style base emits moment_base + compact snippets", async () => { + const client = await connectClient(new StubSource()); + const res = await client.callTool({ + name: "search_transcripts", + arguments: { query: "coffee", link_style: "base", limit: 20 }, + }); + const out = firstText(res); + assert.match(out, /- moment_base: https:\/\/example\.test\/a1\?t=$/m); + assert.ok(out.includes(" - [0:10|10] i love coffee"), out); + assert.ok(!out.includes("](http"), "no full inline links in base snippets"); + assert.match(out, /<moment_base><seconds>/, "the expansion note is present"); + await client.close(); +}); + +test("server: search_transcripts worklist trims the source line (and no moment_base)", async () => { + const client = await connectClient(new StubSource()); + const withSnips = firstText( + await client.callTool({ + name: "search_transcripts", + arguments: { query: "coffee", limit: 5 }, + }), + ); + assert.match(withSnips, /- source: /, "source line present by default"); + + const worklist = firstText( + await client.callTool({ + name: "search_transcripts", + arguments: { + query: "coffee", + limit: 5, + include_snippets: false, + link_style: "base", + }, + }), + ); + assert.ok(!worklist.includes("- source:"), "worklist mode drops the source line"); + assert.ok(!worklist.includes("moment_base"), "…and emits no moment_base either"); + await client.close(); +}); + +test("server: the sweep prompt drives dumb extractors on the cheap model", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { query: "k cups", channel: "chan-a" }, + }); + const text = (got.messages[0].content as { text: string }).text; + assert.match(text, /haiku/, "the default parse model is requested"); + assert.match(text, /DUMB EXTRACTOR/); + assert.match(text, /link_style/); + assert.match(text, /moment_base/); + assert.match(text, /\[mm:ss\|seconds\]/); + assert.match(text, /40 lines/, "the extractor's hard line budget"); + assert.match(text, /<moment_base><seconds>/, "the expansion rule is spelled out"); + await client.close(); +}); + +test("server: the sweep prompt honors parse_model", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { query: "k cups", channel: "chan-a", parse_model: "sonnet" }, + }); + const text = (got.messages[0].content as { text: string }).text; + assert.match(text, /sonnet/); + assert.ok(!/haiku/.test(text), "the default model name is fully replaced"); + await client.close(); +}); diff --git a/mcp/src/search.ts b/mcp/src/search.ts @@ -1,5 +1,16 @@ import { formatDuration } from "yt-dlp-transcript-common/lib/format"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; +import type { Platform } from "yt-dlp-transcript-common/lib/platform"; +import { + forEachLeaf, + isLeaf, + isLeafActive, + isNodeActive, + type GroupNode, + type LayerScope, + type QueryNode, +} from "yt-dlp-transcript-common/lib/searchQuery"; import { matchAliases, type SearchAlias, @@ -14,7 +25,7 @@ import { resolveChannelGroupId, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; -import type { ChannelRef, ShardSource } from "./source"; +import type { ChannelRef, ShardSource, VideoAvailability } from "./source"; export type Snippet = { clock: string; seconds: number; text: string }; @@ -126,9 +137,17 @@ export async function resolveSelectedChannels( export type SearchHit = { videoId: string; + // The video's slug (`<channelSlug>/<id>`) — the archilyzer viewer's `?v=` + // value, used to build a moment deep link (momentUrl). + slug: string; channelSlug: string; channelName: string; siteTitle?: string; + // The public origin of the site that owns this video (a member url in hub + // mode, the site base for a single remote). Used with `slug` for a viewer + // link; absent for a local source (→ platform-link fallback). + siteUrl?: string; + platform?: Platform; title: string; uploadDate: string; webpageUrl?: string; @@ -325,9 +344,12 @@ export async function searchTranscripts( if (matches === 0 && !titleHit) continue; all.push({ videoId: rec.id, + slug: rec.slug, channelSlug: ch.slug, channelName: ch.name, ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}), + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(rec.platform ? { platform: rec.platform } : {}), title: rec.title, uploadDate: rec.uploadDate, webpageUrl: rec.webpageUrl, @@ -376,6 +398,14 @@ export function getWindowedTranscript( after?: number; maxCues?: number; timestamps?: boolean; + // Optional formatter for the bracketed stamp's CONTENTS (e.g. an inline + // Markdown link "[m:ss](url)" or the compact base form "m:ss|156"), given + // the line's clock + start seconds. Bare clock when omitted. + stamp?: (clock: string, seconds: number) => string; + // Cap on merged excerpt lines emitted for this video (default + // WINDOW_LINE_CAP). The earliest lines are kept; `maxCues` above stays the + // per-window bound. + maxLines?: number; } = {}, ): { lines: string[]; matchCount: number } { const cues = record.cues ?? []; @@ -392,11 +422,13 @@ export function getWindowedTranscript( maxCues: opts.maxCues, }), ); - merged = mergeSnippets(merged, win, WINDOW_LINE_CAP); + merged = mergeSnippets(merged, win, opts.maxLines ?? WINDOW_LINE_CAP); } - const lines = merged.map((s) => - timestamps ? `[${s.clock}] ${s.text}` : s.text, - ); + const lines = merged.map((s) => { + if (!timestamps) return s.text; + const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock; + return `[${stamp}] ${s.text}`; + }); return { lines, matchCount }; } @@ -445,3 +477,419 @@ export async function findVideo( } return null; } + +// ─── Full-fidelity spec engine: query tree + filter predicate ─── +// +// `runSearchSpec` is the per-record evaluator that backs the `open_link` / +// `sweep link=` flows: it honors everything an archilyzer share link can carry +// (a composite `qt=` query tree over any mix of scopes, plus the `fc/ft/fa/fav/ +// fdf/fdt` filters), while keeping the same paging / total / truncation +// contract as `searchTranscripts`. The plain `search_transcripts` path stays on +// the simpler `searchTranscripts` scanner above (a trivial single-leaf query), +// so today's callers/tests are unaffected. +// +// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate +// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean +// instead of over slug sets — and the per-scope text extraction of +// searchPipeline.ts. + +// The positive share-filter selection (parseShareV1), as a predicate input. +export type SearchFilters = { + // ft — video / livestream types kept. + videos: boolean; + livestreams: boolean; + // fa — all-ages / age-restricted kept. + allAges: boolean; + restricted: boolean; + // fav — availability three-way (available / unlisted / deleted) kept. + available: boolean; + unlisted: boolean; + deleted: boolean; + // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD". + dateFrom?: string; + dateTo?: string; +}; + +export type SearchSpec = { + // The root of the composite query (a `qt=` tree, or a synthesized single leaf + // for a legacy `q`/`re` link). + tree: GroupNode; + // The decoded share filters, or null for an unfiltered search. + filters?: SearchFilters | null; + // Curated aliases to expand transcripts-scope leaves through (as in the plain + // path). Empty/omitted → no expansion. + aliases?: SearchAlias[]; + // Expand transcripts leaves via aliases (default true; skipped per-leaf for a + // regex leaf). + useAliases?: boolean; +}; + +// A snippet tagged with the scope it came from, carrying the seconds needed to +// build a moment link. `seconds` is 0 for the non-timed scopes (metadata / +// description / tags). +export type ScopedSnippet = { + scope: LayerScope; + track?: string; + clock: string; + seconds: number; + text: string; +}; + +export type SpecHit = { + videoId: string; + slug: string; + channelSlug: string; + channelName: string; + siteTitle?: string; + siteUrl?: string; + platform?: Platform; + title: string; + uploadDate: string; + webpageUrl?: string; + matches: number; + snippets: ScopedSnippet[]; +}; + +export type SpecResult = { + hits: SpecHit[]; + total: number; + offset: number; + limit: number; + hasMore: boolean; + firedAliases: SearchAlias[]; + scanned: { channels: number; pages: number }; + truncated: boolean; +}; + +type LeafMatcher = { scope: LayerScope; test: Matcher }; + +// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless +// they're regex); every other scope matches plain-substring / regex only. Fired +// aliases are unioned for the caller to report. +function buildLeafMatchers( + root: QueryNode, + aliases: SearchAlias[], + useAliases: boolean, +): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } { + const matchers = new Map<string, LeafMatcher>(); + const fired: SearchAlias[] = []; + forEachLeaf(root, (leaf) => { + if (!isLeafActive(leaf)) return; + const aliasAware = + leaf.scope === "transcripts" && !leaf.useRegex && useAliases; + const built = buildMatcher({ + query: leaf.query, + regex: leaf.useRegex, + useAliases: aliasAware, + aliases: aliasAware ? aliases : [], + }); + matchers.set(leaf.id, { scope: leaf.scope, test: built.match }); + for (const a of built.firedAliases) { + if (!fired.some((x) => x.id === a.id)) fired.push(a); + } + }); + return { matchers, fired }; +} + +// Per-record context the tree evaluates against. +type RecordCtx = { + title: string; + channel: string; + description: string; + tags: string; + cues: Cue[]; + chatCues: Cue[]; + snippetsPerVideo: number; + includeSnippets: boolean; +}; + +type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] }; + +function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome { + if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] }; + const hits: ScopedSnippet[] = []; + const push = (s: ScopedSnippet): void => { + if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s); + }; + let count = 0; + switch (m.scope) { + case "transcripts": + for (const cue of ctx.cues) { + if (!m.test(cue.text)) continue; + count++; + push({ + scope: "transcripts", + clock: clock(cue.start), + seconds: cue.start, + text: truncate(cue.text), + }); + } + break; + case "chat": + for (const cue of ctx.chatCues) { + if (!m.test(cue.text)) continue; + count++; + push({ + scope: "chat", + track: "live_chat", + clock: clock(cue.start), + seconds: cue.start, + text: truncate(cue.text), + }); + } + break; + case "metadata": { + const titleHit = m.test(ctx.title); + const channelHit = m.test(ctx.channel); + if (titleHit || channelHit) { + count++; + if (titleHit) { + push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) }); + } else { + push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` }); + } + } + break; + } + case "description": + if (ctx.description && m.test(ctx.description)) { + count++; + push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) }); + } + break; + case "tags": + if (ctx.tags && m.test(ctx.tags)) { + count++; + push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) }); + } + break; + } + return { matched: count > 0, count, hits }; +} + +type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] }; + +// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate, +// per record. A negated node contributes no hits (like the browser's `diff`). +// An inactive (empty) subtree is identity (matches, no hits). +function evalNode( + node: QueryNode, + matchers: Map<string, LeafMatcher>, + ctx: RecordCtx, +): NodeOutcome { + if (!isNodeActive(node)) return { match: true, count: 0, hits: [] }; + if (isLeaf(node)) { + const m = matchers.get(node.id); + if (!m) return { match: true, count: 0, hits: [] }; + const r = evalLeaf(node, m, ctx); + if (node.negate) return { match: !r.matched, count: 0, hits: [] }; + return { + match: r.matched, + count: node.contributeHits ? r.count : 0, + hits: node.contributeHits ? r.hits : [], + }; + } + const active = node.children.filter(isNodeActive); + if (active.length === 0) return { match: true, count: 0, hits: [] }; + if (node.op === "AND") { + let allMatch = true; + let count = 0; + const hits: ScopedSnippet[] = []; + for (const c of active) { + const r = evalNode(c, matchers, ctx); + if (!r.match) { + allMatch = false; + break; + } + count += r.count; + hits.push(...r.hits); + } + const match = node.negate ? !allMatch : allMatch; + return match && !node.negate + ? { match, count, hits } + : { match, count: 0, hits: [] }; + } + // OR + let any = false; + let count = 0; + const hits: ScopedSnippet[] = []; + for (const c of active) { + const r = evalNode(c, matchers, ctx); + if (r.match) { + any = true; + count += r.count; + hits.push(...r.hits); + } + } + const match = node.negate ? !any : any; + return match && !node.negate + ? { match, count, hits } + : { match, count: 0, hits: [] }; +} + +// The share-filter predicate over a transcript record + its availability. +// Mirrors SearchSessionContext.tsx `passesFilter` exactly. +function passesFilters( + rec: TranscriptDetail, + f: SearchFilters, + avail: VideoAvailability | undefined, +): boolean { + // ft — type + if (rec.isLivestream ? !f.livestreams : !f.videos) return false; + // fa — audience + if (rec.ageRestricted ? !f.restricted : !f.allAges) return false; + // fav — availability three-way + if (avail?.isDeleted) { + if (!f.deleted) return false; + } else if (avail?.isUnlisted) { + if (!f.unlisted) return false; + } else if (!f.available) { + return false; + } + // fdf / fdt — upload-date range (lexicographic on YYYYMMDD) + if (f.dateFrom && rec.uploadDate < f.dateFrom) return false; + if (f.dateTo && rec.uploadDate > f.dateTo) return false; + return true; +} + +// True when the fav filter could exclude something (so availability must be +// fetched). If every availability bucket is kept, there's nothing to look up. +function needsAvailability(f: SearchFilters | null | undefined): boolean { + return !!f && (!f.available || !f.unlisted || !f.deleted); +} + +// A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a +// chat-scope leaf only pulls the subs shards it actually touches. +function makeChatFetcher(source: ShardSource) { + const manifests = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsManifest"]>>>>(); + const pages = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsPage"]>>>>(); + return async function chatCues(ch: ChannelRef, rec: TranscriptDetail): Promise<Cue[]> { + let mp = manifests.get(ch.key); + if (!mp) { + mp = source.subsManifest(ch).catch(() => null); + manifests.set(ch.key, mp); + } + const manifest = await mp; + if (!manifest) return []; + const page = manifest.slugToPage[rec.id]; + if (page === undefined) return []; + const pk = `${ch.key}:${page}`; + let pp = pages.get(pk); + if (!pp) { + pp = source.subsPage(ch, page).catch(() => []); + pages.set(pk, pp); + } + const records = await pp; + const found = records.find((r) => r.slug === rec.slug || r.id === rec.id); + const lc = found?.tracks?.live_chat; + return Array.isArray(lc) ? lc : []; + }; +} + +// Run a full-fidelity spec over the given (already-resolved) channel set. Same +// paging/total/truncation semantics as searchTranscripts. +export async function runSearchSpec( + source: ShardSource, + channels: ChannelRef[], + spec: SearchSpec, + opts: { + offset?: number; + limit?: number; + includeSnippets?: boolean; + maxPages?: number; + snippetsPerVideo?: number; + } = {}, +): Promise<SpecResult> { + const limit = opts.limit ?? 20; + const offset = Math.max(0, opts.offset ?? 0); + const maxPages = opts.maxPages ?? MAX_PAGES; + const includeSnippets = opts.includeSnippets !== false; + const snippetsPerVideo = opts.snippetsPerVideo ?? 4; + const filters = spec.filters ?? null; + + const { matchers, fired } = buildLeafMatchers( + spec.tree, + spec.aliases ?? [], + spec.useAliases !== false, + ); + const wantsChat = [...matchers.values()].some((m) => m.scope === "chat"); + const chatCuesFor = wantsChat ? makeChatFetcher(source) : null; + const availability = needsAvailability(filters) + ? await source.availabilityMap() + : null; + + const all: SpecHit[] = []; + let pagesScanned = 0; + let channelsScanned = 0; + let truncated = false; + + outer: for (const ch of channels) { + let manifest; + try { + manifest = await source.transcriptsManifest(ch); + } catch { + continue; + } + channelsScanned++; + for (let page = 0; page < manifest.pageCount; page++) { + if (pagesScanned >= maxPages) { + truncated = true; + break outer; + } + let records: TranscriptDetail[]; + try { + records = await source.transcriptPage(ch, page); + } catch { + continue; + } + pagesScanned++; + for (const rec of records) { + if (filters && !passesFilters(rec, filters, availability?.get(rec.slug))) { + continue; + } + const ctx: RecordCtx = { + title: rec.title ?? "", + channel: rec.channel ?? ch.name, + description: rec.description ?? "", + tags: (rec.tags ?? []).join(", "), + cues: rec.cues ?? [], + chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [], + snippetsPerVideo, + includeSnippets, + }; + const r = evalNode(spec.tree, matchers, ctx); + if (!r.match) continue; + all.push({ + videoId: rec.id, + slug: rec.slug, + channelSlug: ch.slug, + channelName: ch.name, + ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}), + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(rec.platform ? { platform: rec.platform } : {}), + title: rec.title, + uploadDate: rec.uploadDate, + webpageUrl: rec.webpageUrl, + matches: r.count || 1, + snippets: r.hits, + }); + if (all.length >= HARD_VIDEO_CAP) { + truncated = true; + break outer; + } + } + } + } + + const total = all.length; + return { + hits: all.slice(offset, offset + limit), + total, + offset, + limit, + hasMore: offset + limit < total, + firedAliases: fired, + scanned: { channels: channelsScanned, pages: pagesScanned }, + truncated, + }; +} diff --git a/mcp/src/server.ts b/mcp/src/server.ts @@ -7,6 +7,8 @@ import { } from "@modelcontextprotocol/sdk/types.js"; import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown"; import { formatDate } from "yt-dlp-transcript-common/lib/format"; +import { momentUrl, momentBaseUrl } from "yt-dlp-transcript-common/lib/momentUrl"; +import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { sortGroups, @@ -14,7 +16,13 @@ import { FALLBACK_GROUP, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; -import { HubSource, type ChannelRef, type HubSite, type ShardSource } from "./source"; +import { + HubSource, + RemoteSource, + type ChannelRef, + type HubSite, + type ShardSource, +} from "./source"; import type { SourceSpec } from "./sources"; import type { SourceController } from "./sourceController"; import { @@ -22,8 +30,89 @@ import { findVideo, buildMatcher, getWindowedTranscript, + runSearchSpec, type SearchResult, + type SpecHit, + type ScopedSnippet, } from "./search"; +import { + decodeShareLink, + applyLinkOverrides, + renderQueryTree, + describeFilters, + type DecodedLink, + type LinkOverrides, +} from "./shareLink"; + +// The building blocks momentUrl needs from a hit (viewer origin comes from the +// per-video siteUrl, else the source's single public origin). +type LinkableHit = { + slug: string; + siteUrl?: string; + webpageUrl?: string; + platform?: Platform; +}; + +// A moment deep link for one cited second of a hit, or null when neither a +// viewer nor a platform link can be built (e.g. a local source + no webpage url). +function momentLinkFor( + source: ShardSource, + h: LinkableHit, + seconds: number, +): string | null { + return momentUrl({ + siteOrigin: h.siteUrl ?? source.publicOrigin(), + slug: h.slug, + seconds, + webpageUrl: h.webpageUrl, + platform: h.platform, + }); +} + +// A `[m:ss]` timestamp rendered as a Markdown link when we can build one, else +// bare. Used inside a `- [ … ] text` snippet line → `- [[m:ss](url)] text`. +function stampMarkup( + source: ShardSource, + h: LinkableHit, + clock: string, + seconds: number, +): string { + const url = momentLinkFor(source, h, seconds); + return url ? `[${clock}](${url})` : clock; +} + +// The requested timestamp-link form: "base" is the compact agent-pipeline form +// (one `moment_base` per video + `[m:ss|seconds]` lines); anything else — the +// default — is the full inline Markdown links. +function linkStyleOf(args: Record<string, unknown>): "inline" | "base" { + return args.link_style === "base" ? "base" : "inline"; +} + +// The compact stamp contents for link_style:"base": `m:ss|156`. The integer is +// floored exactly like momentUrl floors its seconds, so appending it to the +// video's moment_base cites the identical second the inline link would. +function baseStamp(clock: string, seconds: number): string { + return `${clock}|${Math.floor(seconds)}`; +} + +// The appendable moment base for a hit (mirrors momentLinkFor), or null when +// none can be built — no viewer origin AND the platform's time param doesn't +// take raw seconds (Twitch) or doesn't exist (Rumble/Kick/no URL). +function momentBaseFor(source: ShardSource, h: LinkableHit): string | null { + return momentBaseUrl({ + siteOrigin: h.siteUrl ?? source.publicOrigin(), + slug: h.slug, + webpageUrl: h.webpageUrl, + platform: h.platform, + }); +} + +// Footer note stating the expansion rule for link_style:"base" output. +const BASE_EXPANSION_NOTE = + "link_style base: full moment link = `<moment_base><seconds>` — append the " + + "integer after the `|` to that video's moment_base, e.g. `[title @ 2:36]" + + "(<moment_base>156)`; a video with no moment_base line → cite its source " + + "url plain"; type ToolResult = { content: { type: "text"; text: string }[]; @@ -127,6 +216,18 @@ const TOOLS = [ "Max shard pages to scan before stopping (default 400). Reaching " + "it marks coverage partial.", }, + link_style: { + type: "string", + enum: ["inline", "base"], + description: + "Timestamp link form (default 'inline': every [m:ss] is a full " + + "Markdown moment link). 'base' is a compact agent-pipeline form: " + + "each video gets one '- moment_base:' header line (a URL ending " + + "in 't=') and snippet stamps become [m:ss|<seconds>]; expand to a " + + "full link by appending the integer seconds to the moment_base " + + "([title @ m:ss](<moment_base><seconds>)). A video with no " + + "buildable base omits the line — cite its source url plain.", + }, }, required: ["query"], additionalProperties: false, @@ -209,6 +310,25 @@ const TOOLS = [ type: "number", description: "Seconds of context after each match (default 30).", }, + link_style: { + type: "string", + enum: ["inline", "base"], + description: + "Timestamp link form (default 'inline': every [m:ss] is a full " + + "Markdown moment link). 'base' is a compact agent-pipeline form: " + + "each video gets one '- moment_base:' header line (a URL ending " + + "in 't=') and line stamps become [m:ss|<seconds>]; expand to a " + + "full link by appending the integer seconds to the moment_base " + + "([title @ m:ss](<moment_base><seconds>)). A video with no " + + "buildable base omits the line — cite its source url plain.", + }, + max_lines: { + type: "number", + description: + "Cap on merged excerpt lines per video when a query is given " + + "(default 200; the earliest lines are kept). Ignored without a " + + "query.", + }, }, required: ["video_ids"], additionalProperties: false, @@ -291,6 +411,103 @@ const TOOLS = [ "--local flag or env) and clear the persisted selection.", inputSchema: { type: "object", properties: {}, additionalProperties: false }, }, + { + name: "open_link", + description: + "Paste an archilyzer viewer **share link** (origin + a `qt=` query tree + " + + "`fc/ft/fa/fav/fdf/fdt` filters) to re-run that exact search here — at full " + + "fidelity, with clickable timestamped result links. Two-phase: by default " + + "(apply:false) it PREVIEWS — decodes the link and returns a plan (resolved " + + "source, the query tree rendered readably, every active filter, the channel " + + "scope validated against the live corpus, and anything ignored, e.g. the " + + "vestigial `fk` tracks) WITHOUT switching source or searching. Adjust in " + + "natural language by re-calling with `overrides` (e.g. clear_availability " + + "to drop the availability filter, channels to re-scope, query to replace " + + "the search). When the plan looks right, call again with apply:true: it " + + "switches the active source to the link's origin (hub or single-site, " + + "auto-detected) and returns the first page of linked results. Read-only.", + inputSchema: { + type: "object", + properties: { + link: { + type: "string", + description: "The archilyzer viewer share URL to decode and run.", + }, + apply: { + type: "boolean", + description: + "false (default) previews the plan only; true switches source and " + + "runs the search.", + }, + overrides: { + type: "object", + description: + "Structured edits to the decoded search, so a natural-language " + + "adjustment can be honored by re-calling with the changed facet.", + properties: { + clear_filters: { type: "boolean", description: "Drop all filters." }, + clear_availability: { + type: "boolean", + description: "Remove the availability (deleted/unlisted) filter.", + }, + clear_type: { + type: "boolean", + description: "Remove the video/livestream type filter.", + }, + clear_age: { + type: "boolean", + description: "Remove the all-ages/age-restricted filter.", + }, + clear_dates: { + type: "boolean", + description: "Remove the upload-date range filter.", + }, + clear_channels: { + type: "boolean", + description: "Widen the channel scope to the whole corpus.", + }, + channels: { + type: "array", + items: { type: "string" }, + description: "Replace the channel scope with these channel names.", + }, + date_from: { + type: "string", + description: "Set the inclusive lower upload-date bound (YYYYMMDD).", + }, + date_to: { + type: "string", + description: "Set the inclusive upper upload-date bound (YYYYMMDD).", + }, + query: { + type: "string", + description: "Replace the whole query with this single term.", + }, + regex: { + type: "boolean", + description: "Treat the override query as a regex.", + }, + query_scope: { + type: "string", + enum: ["transcripts", "chat", "metadata", "description", "tags"], + description: "Scope for the override query (default transcripts).", + }, + }, + additionalProperties: false, + }, + limit: { + type: "number", + description: "Results per page when apply:true (default 20).", + }, + offset: { + type: "number", + description: "Skip this many matches when apply:true (default 0).", + }, + }, + required: ["link"], + additionalProperties: false, + }, + }, ]; // The slice of SourceController the server needs. A bare ShardSource is wrapped @@ -378,6 +595,8 @@ export function createServer(sourceOrController: ShardSource | SourceController) return await handleUseSource(controller, args); case "reset_source": return await handleResetSource(controller); + case "open_link": + return await handleOpenLink(controller, args); default: return errorText(`unknown tool: ${name}`); } @@ -465,6 +684,8 @@ async function handleSearch( ): Promise<ToolResult> { const query = String(args.query ?? "").trim(); if (!query) return errorText("query is required"); + const includeSnippets = args.include_snippets !== false; + const base = linkStyleOf(args) === "base"; const result = await searchTranscripts(source, { query, channel: typeof args.channel === "string" ? args.channel : undefined, @@ -474,7 +695,7 @@ async function handleSearch( regex: args.regex === true, limit: typeof args.limit === "number" ? args.limit : undefined, offset: typeof args.offset === "number" ? args.offset : undefined, - includeSnippets: args.include_snippets !== false, + includeSnippets, useAliases: args.use_aliases !== false, maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined, }); @@ -502,17 +723,29 @@ async function handleSearch( return text(`${head}${footer}`); } const blocks = result.hits.map((h) => { + // Worklist mode (include_snippets:false) trims the source line too — the + // caller only wants ids/titles/counts, so no per-video URLs at all. + const baseUrl = base && includeSnippets ? momentBaseFor(source, h) : null; const head = `### ${h.title}\n` + `- video_id: ${h.videoId} | channel: ${h.channelName}` + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + - (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); - const snips = h.snippets.map((s) => ` - [${s.clock}] ${s.text}`).join("\n"); + (includeSnippets && h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "") + + (baseUrl ? `\n- moment_base: ${baseUrl}` : ""); + const snips = h.snippets + .map((s) => + base + ? ` - [${baseStamp(s.clock, s.seconds)}] ${s.text}` + : ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`, + ) + .join("\n"); return snips ? `${head}\n${snips}` : head; }); + // Worklist mode has no stamps (or bases) to expand — skip the note too. + const baseNote = base && includeSnippets ? `\n\n(${BASE_EXPANSION_NOTE})` : ""; return text( - `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}`, + `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}${baseNote}`, ); } @@ -594,6 +827,11 @@ async function handleGetTranscripts( const before = typeof args.before === "number" ? args.before : 30; const after = typeof args.after === "number" ? args.after : 30; + const base = linkStyleOf(args) === "base"; + const maxLinesArg = + typeof args.max_lines === "number" ? Math.floor(args.max_lines) : undefined; + const maxLines = + maxLinesArg !== undefined && maxLinesArg >= 1 ? maxLinesArg : undefined; const blocks: string[] = []; const missing: string[] = []; @@ -604,18 +842,36 @@ async function handleGetTranscripts( continue; } const { record, ch } = found; + const link: LinkableHit = { + slug: record.slug, + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), + ...(record.platform ? { platform: record.platform } : {}), + }; + const baseUrl = base ? momentBaseFor(source, link) : null; + // The bracketed stamp's contents: the full inline Markdown link (byte- + // identical to the pre-link_style output), or the compact base form. + const stamp = base + ? baseStamp + : (clock: string, seconds: number): string => { + const url = momentLinkFor(source, link, seconds); + return url ? `[${clock}](${url})` : clock; + }; const head = `## ${record.title || id}\n` + `- video_id: ${id} | channel: ${ch.name}` + (ch.siteTitle ? ` | site: ${ch.siteTitle}` : "") + ` | uploaded: ${formatDate(record.uploadDate)}` + - (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : ""); + (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "") + + (baseUrl ? `\n- moment_base: ${baseUrl}` : ""); if (matcher) { const { lines, matchCount } = getWindowedTranscript(record, matcher.match, { before, after, timestamps, + stamp, + maxLines, }); const body = matchCount === 0 @@ -623,10 +879,21 @@ async function handleGetTranscripts( : `_(${matchCount} matching line(s), windowed)_\n${lines.join("\n")}`; blocks.push(`${head}\n\n${body}`); } else { - const md = transcriptToMarkdown(record, { - timestamps, - includeTags: true, - }); + const md = transcriptToMarkdown( + record, + base + ? { + timestamps, + includeTags: true, + stampForCue: baseStamp, + extraMeta: baseUrl ? [`moment_base: ${baseUrl}`] : undefined, + } + : { + timestamps, + includeTags: true, + linkForCue: (seconds) => momentLinkFor(source, link, seconds), + }, + ); blocks.push(md.trim()); } } @@ -638,6 +905,7 @@ async function handleGetTranscripts( } if (missing.length > 0) notes.push(`not found: ${missing.join(", ")}`); if (dropped > 0) notes.push(`${dropped} extra id(s) beyond the 20-cap dropped`); + if (base) notes.push(BASE_EXPANSION_NOTE); const footer = notes.length > 0 ? `\n\n(${notes.join("; ")})` : ""; if (blocks.length === 0) { @@ -658,9 +926,17 @@ async function handleGetTranscript( typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`video not found: ${videoId}`); - const md = transcriptToMarkdown(found.record, { + const { record, ch } = found; + const link: LinkableHit = { + slug: record.slug, + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), + ...(record.platform ? { platform: record.platform } : {}), + }; + const md = transcriptToMarkdown(record, { timestamps: args.timestamps !== false, includeTags: true, + linkForCue: (seconds) => momentLinkFor(source, link, seconds), }); return text(md); } @@ -896,6 +1172,273 @@ async function handleResetSource( ); } +// ─── open_link: paste a share link → preview, adjust, apply ─── + +// Map the snake_cased tool `overrides` object onto the LinkOverrides shape. +function parseOverrides(raw: unknown): LinkOverrides | undefined { + if (!raw || typeof raw !== "object") return undefined; + const o = raw as Record<string, unknown>; + const bool = (k: string): boolean | undefined => + o[k] === true ? true : undefined; + const str = (k: string): string | undefined => + typeof o[k] === "string" ? (o[k] as string) : undefined; + const scope = str("query_scope"); + return { + clearFilters: bool("clear_filters"), + clearAvailability: bool("clear_availability"), + clearType: bool("clear_type"), + clearAge: bool("clear_age"), + clearDates: bool("clear_dates"), + clearChannels: bool("clear_channels"), + channels: strArray(o.channels), + dateFrom: str("date_from"), + dateTo: str("date_to"), + query: str("query"), + regex: o.regex === true ? true : undefined, + queryScope: + scope === "transcripts" || + scope === "chat" || + scope === "metadata" || + scope === "description" || + scope === "tags" + ? scope + : undefined, + }; +} + +type OriginProbe = { kind: "hub" | "remote"; siteCount?: number; channelCount?: number }; + +// Probe an origin's corpus.json to classify it as a federated hub (has a +// `sites[]` array) or a single site (has `channels[]`). Transient — no source is +// committed. Defaults to single-site when the probe can't be read. +async function probeOrigin(origin: string): Promise<OriginProbe> { + try { + const res = await fetch(`${origin}/corpus.json`); + if (res.ok) { + const j = (await res.json()) as { + sites?: unknown[]; + channels?: unknown[]; + }; + if (Array.isArray(j.sites)) return { kind: "hub", siteCount: j.sites.length }; + if (Array.isArray(j.channels)) { + return { kind: "remote", channelCount: j.channels.length }; + } + } + } catch { + // unreachable — fall through to the single-site default + } + return { kind: "remote" }; +} + +type ChannelScope = { + channels: ChannelRef[]; + all: boolean; + matched: string[]; + unknown: string[]; +}; + +// Resolve a decoded link's `fc` channel NAMES against a live source's channels +// (by display name or slug, case-insensitive). No channel filter → whole corpus. +async function resolveLinkChannels( + source: ShardSource, + decoded: DecodedLink, +): Promise<ChannelScope> { + const all = await source.listChannels(); + if (!decoded.hasChannelFilter) { + return { channels: all, all: true, matched: [], unknown: [] }; + } + const want = decoded.channelNames.map((n) => n.toLowerCase()); + const matches = (c: ChannelRef): boolean => + want.includes(c.name.toLowerCase()) || want.includes(c.slug.toLowerCase()); + const channels = all.filter(matches); + const matched = channels.map((c) => c.name); + const unknown = decoded.channelNames.filter( + (n) => + !all.some( + (c) => + c.name.toLowerCase() === n.toLowerCase() || + c.slug.toLowerCase() === n.toLowerCase(), + ), + ); + return { channels, all: false, matched, unknown }; +} + +// Render the human-readable preview/echo plan for a decoded link. +function buildLinkPlan( + decoded: DecodedLink, + probe: OriginProbe, + scope: ChannelScope, + applied: boolean, +): string { + const lines: string[] = []; + lines.push(applied ? "## Applied share link" : "## Share-link preview"); + const kindNote = + probe.kind === "hub" + ? `hub (${probe.siteCount ?? "?"} member site(s))` + : `single site${probe.channelCount != null ? ` (${probe.channelCount} channel(s))` : ""}`; + lines.push(`- Source: ${decoded.origin} — ${kindNote}`); + lines.push(`- Query: ${renderQueryTree(decoded.tree)} _(from ${decoded.querySource})_`); + + if (scope.all) { + lines.push(`- Scope: whole corpus`); + } else { + const names = scope.matched.length ? scope.matched.join(", ") : "(none)"; + lines.push(`- Scope: ${scope.channels.length} channel(s): ${names}`); + if (scope.unknown.length > 0) { + lines.push(` - ⚠ channel name(s) not found in this corpus: ${scope.unknown.join(", ")}`); + } + if (scope.channels.length === 0) { + lines.push( + ` - ⚠ the link selects 0 channels here — nothing will match. Re-call ` + + `with overrides.clear_channels to search the whole corpus.`, + ); + } + } + + const filterLines = describeFilters(decoded.filters); + if (filterLines.length > 0) { + lines.push(`- Filters:`); + for (const f of filterLines) lines.push(` - ${f}`); + } else { + lines.push(`- Filters: none`); + } + + for (const w of decoded.warnings) lines.push(`- Note: ${w}`); + + if (!applied) { + lines.push( + ``, + `No search run yet. Call again with apply:true to switch source and ` + + `search, or pass overrides to adjust first.`, + ); + } + return lines.join("\n"); +} + +// A ScopedSnippet line: a linked `[m:ss]` for timed scopes, or a scope-tagged +// note for the non-timed scopes (metadata / description / tags). +function renderScopedSnippet( + source: ShardSource, + hit: SpecHit, + s: ScopedSnippet, +): string { + if (s.seconds > 0) { + const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `; + return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`; + } + return ` - [${s.scope}] ${s.text}`; +} + +// open_link results (and get_transcript) stay inline-linked for now — a +// link_style:"base" form for them is future work. +function renderSpecResults(source: ShardSource, hits: SpecHit[]): string { + return hits + .map((h) => { + const head = + `### ${h.title}\n` + + `- video_id: ${h.videoId} | channel: ${h.channelName}` + + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + + (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); + const snips = h.snippets + .map((s) => renderScopedSnippet(source, h, s)) + .join("\n"); + return snips ? `${head}\n${snips}` : head; + }) + .join("\n\n"); +} + +async function handleOpenLink( + controller: SourceControllerLike, + args: Record<string, unknown>, +): Promise<ToolResult> { + const linkStr = typeof args.link === "string" ? args.link.trim() : ""; + if (!linkStr) return errorText("link is required"); + + let decoded: DecodedLink; + try { + decoded = decodeShareLink(linkStr); + } catch (e) { + return errorText(`could not parse link as a URL: ${(e as Error).message}`); + } + decoded = applyLinkOverrides(decoded, parseOverrides(args.overrides)); + + const apply = args.apply === true; + const probe = await probeOrigin(decoded.origin); + + if (!apply) { + // Preview: resolve channels against a transient source; do not commit. + const preview: ShardSource = + probe.kind === "hub" + ? new HubSource(decoded.origin) + : new RemoteSource(decoded.origin); + let scope: ChannelScope; + try { + scope = await resolveLinkChannels(preview, decoded); + } catch (e) { + return errorText( + `could not read the corpus at ${decoded.origin}: ${(e as Error).message}`, + ); + } + return text(buildLinkPlan(decoded, probe, scope, false)); + } + + // Apply: commit the source (reusing the controller), then search. + const spec: SourceSpec = + probe.kind === "hub" + ? { kind: "hub", url: decoded.origin } + : { kind: "remote", url: decoded.origin }; + try { + await controller.switchTo(spec); + } catch (e) { + return errorText(`open_link could not switch source: ${(e as Error).message}`); + } + const source = controller.current; + + let scope: ChannelScope; + try { + scope = await resolveLinkChannels(source, decoded); + } catch (e) { + return errorText( + `switched to ${source.label} but could not read its corpus: ${(e as Error).message}`, + ); + } + const plan = buildLinkPlan(decoded, probe, scope, true); + + const limit = typeof args.limit === "number" ? args.limit : 20; + const offset = typeof args.offset === "number" ? args.offset : 0; + const result = await runSearchSpec( + source, + scope.channels, + { + tree: decoded.tree, + filters: decoded.filters, + aliases: await source.loadAliases(), + }, + { limit, offset }, + ); + + const aliasNote = describeFiredAliases(result.firedAliases); + const rangeStart = result.total === 0 ? 0 : result.offset + 1; + const rangeEnd = result.offset + result.hits.length; + const footer = + `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` + + `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` + + `page(s) across ${result.scanned.channels} channel(s)` + + (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") + + (aliasNote ? `; ${aliasNote}` : "") + + ")"; + + const body = + result.hits.length === 0 + ? result.total === 0 + ? "No matches for this link's search." + : `No matches in this page (offset ${result.offset} is past the ${result.total} total).` + : `${result.total} video(s) matched:\n\n${renderSpecResults(source, result.hits)}`; + + return text(`${plan}\n\n---\n\n${body}${footer}`); +} + // ─── Prompts: the first-class `sweep` entry point ─── // A single slash command that drives Claude Code to run the browser's // "corpus sweep" on plan usage: enumerate a query's full match set, batch the @@ -912,7 +1455,21 @@ const PROMPTS = [ "on plan usage, no API key. With no scope arg it lists the channel groups " + "and asks you to pick a group/channels (or confirm 'all') before sweeping.", arguments: [ - { name: "query", description: "Term or phrase to sweep for.", required: true }, + { + name: "query", + description: + "Term or phrase to sweep for. Optional when a `link` is given (the " + + "link supplies the query).", + required: false, + }, + { + name: "link", + description: + "An archilyzer viewer share URL to seed the sweep from — its origin " + + "(source), query tree, and filters are decoded and applied via " + + "open_link before enumerating.", + required: false, + }, { name: "channel", description: "Optional channel slug/name to restrict the sweep to.", @@ -943,6 +1500,14 @@ const PROMPTS = [ required: false, }, { + name: "parse_model", + description: + "Model to request for the per-batch extractor subagents (default " + + "'haiku'). The extractors only quote verbatim, so the cheapest " + + "model wins; all synthesis stays with the orchestrator.", + required: false, + }, + { name: "report_path", description: "Report file to write (default ./sweep-report.md).", required: false, @@ -958,7 +1523,10 @@ function argStr(args: Record<string, unknown>, key: string): string | undefined function buildSweepPrompt(args: Record<string, unknown>) { const query = argStr(args, "query"); - if (!query) throw new Error("sweep requires a query argument"); + const link = argStr(args, "link"); + if (!query && !link) { + throw new Error("sweep requires a query argument (or a link)"); + } const channel = argStr(args, "channel"); const group = argStr(args, "group"); const channelsRaw = argStr(args, "channels"); @@ -967,11 +1535,13 @@ function buildSweepPrompt(args: Record<string, unknown>) { : []; const directive = argStr(args, "directive") ?? "key claims & contradictions"; const batchSize = argStr(args, "batch_size") ?? "8"; + const parseModel = argStr(args, "parse_model") ?? "haiku"; const reportPath = argStr(args, "report_path") ?? "./sweep-report.md"; + const subject = query ? `"${query}"` : "the share link's search"; // Human-readable scope clauses + the literal search_transcripts scope args to - // pass. When none is given, the sweep must pick-first (list + ask) rather than - // silently scanning the whole corpus. + // pass. When none is given (and there's no link), the sweep must pick-first + // (list + ask) rather than silently scanning the whole corpus. const scopeClauses: string[] = []; if (channel) scopeClauses.push(`channel "${channel}"`); if (channels.length > 0) @@ -980,70 +1550,117 @@ function buildSweepPrompt(args: Record<string, unknown>) { const hasScope = scopeClauses.length > 0; const scopeArgsText = scopeClauses.join(" and "); - const introScope = hasScope - ? ` scoped to ${scopeArgsText}` - : " over a scope you will confirm with me first (see step 1)"; + const introScope = link + ? ` seeded from a share link (its source, query tree, and filters — see step 1)` + : hasScope + ? ` scoped to ${scopeArgsText}` + : " over a scope you will confirm with me first (see step 1)"; - // Build the numbered steps. With an explicit scope we search directly; with - // none we insert a pick-first step and the Search step uses the chosen scope. const steps: string[] = []; - if (!hasScope) { + // ── Discovery / enumeration differs for a link-seeded sweep vs a query one ── + if (link) { + steps.push( + `**Decode & confirm the link.** Call \`open_link\` with ` + + `link="${link}" (preview mode — no apply). It returns a plan: the ` + + `resolved source (origin, hub or single-site), the query tree, every ` + + `active filter, the channel scope validated against the corpus, and any ` + + `ignored bits (e.g. the vestigial \`fk\` tracks). **Show me the plan and ` + + `confirm it captures what I want.** If I ask for a change ("drop the ` + + `availability filter", "only channel X", "search Y instead"), re-call ` + + `\`open_link\` with the matching \`overrides\` until the plan is right.`, + ); + steps.push( + `**Apply & enumerate.** Call \`open_link\` again with the confirmed ` + + `arguments plus apply:true — this switches the active source to the ` + + `link's origin and runs the search (the full query tree + filters, at ` + + `fidelity). Page it with a rising \`offset\` (offset += limit) until ` + + `\`has_more\` is no to collect the whole worklist of video ids; note the ` + + `\`total\`. Results already carry linked \`[mm:ss](url)\` timestamps and ` + + `the applied plan is echoed at the top — record the source, query, and ` + + `filters in the report. If coverage is PARTIAL, say so.`, + ); + } else { + if (!hasScope) { + steps.push( + `**Choose the scope first — do NOT default to the whole corpus.** No ` + + `channel/channels/group was supplied. Call \`list_channels\`, present ` + + `the channel groups and their channels to me, and ask which group(s) ` + + `or channel(s) to sweep — or to confirm **all** for the whole corpus. ` + + `Wait for my choice before enumerating anything. Only sweep everything ` + + `if I explicitly choose "all". Use my choice as the ` + + `\`channel\`/\`channels\`/\`group\` scope in every ` + + `\`search_transcripts\` call below.`, + ); + } steps.push( - `**Choose the scope first — do NOT default to the whole corpus.** No ` + - `channel/channels/group was supplied. Call \`list_channels\`, present ` + - `the channel groups and their channels to me, and ask which group(s) or ` + - `channel(s) to sweep — or to confirm **all** for the whole corpus. Wait ` + - `for my choice before enumerating anything. Only sweep everything if I ` + - `explicitly choose "all". Use my choice as the \`channel\`/\`channels\`/` + - `\`group\` scope in every \`search_transcripts\` call below.`, + `**Search.** Call \`search_transcripts\` with query "${query}"` + + (hasScope + ? ` and ${scopeArgsText}` + : ` and the scope I chose in step 1`) + + `. Curated aliases auto-expand the query — the footer reports which ` + + `fired (e.g. mis-transcribed spellings) and names the resolved scope ` + + `(and warns about any channel/group token that matched nothing — fix a ` + + `typo before continuing). Treat the *expanded* match set as your target ` + + `and mention the expansion and the scope in the report.`, + ); + steps.push( + `**Enumerate the full worklist.** Page the complete set with ` + + `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` + + `until \`has_more\` is false — this gives you every id/title/channel/date ` + + `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` + + `(page/video cap), say so in the report — the sweep is then a sample, ` + + `not exhaustive.`, ); } steps.push( - `**Search.** Call \`search_transcripts\` with query "${query}"` + - (hasScope - ? ` and ${scopeArgsText}` - : ` and the scope I chose in step 1`) + - `. Curated aliases auto-expand the query — the footer reports which fired ` + - `(e.g. mis-transcribed spellings) and names the resolved scope (and warns ` + - `about any channel/group token that matched nothing — fix a typo before ` + - `continuing). Treat the *expanded* match set as your target and mention ` + - `the expansion and the scope in the report.`, - ); - - steps.push( - `**Enumerate the full worklist.** Page the complete set with ` + - `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` + - `until \`has_more\` is false — this gives you every id/title/channel/date ` + - `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` + - `(page/video cap), say so in the report — the sweep is then a sample, not ` + - `exhaustive.`, - ); - - steps.push( `**Plan.** With N total matches and a batch size of ${batchSize}, that is ` + `\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch ` + `count) before you start.`, ); + // ── Feature 2: dumb-extractor-per-batch map-reduce on the cheapest model — + // the extractor only quotes verbatim base-form excerpts, so the heavy + // transcript text never reaches the orchestrator (and never costs the big + // model). Feature 1: linked citations, expanded from moment_base + seconds. steps.push( - `**Per batch**, for each group of up to ${batchSize} video ids:\n` + - ` - Call \`get_transcripts\` with those ids **and the query** so each ` + - `transcript comes back as bounded, timestamped excerpt windows around the ` + - `matches (alias-correct, high-signal).\n` + - ` - **Cross-reference** the batch against the report so far. Upsert ` + + `**Per batch (map-reduce), for each group of up to ${batchSize} video ids:**\n` + + ` - **Spawn a subagent as a DUMB EXTRACTOR on the cheapest model** — ` + + `use the Task tool and request model "${parseModel}" (its model ` + + `parameter, or a "${parseModel}"-backed agent type); if no model ` + + `override is available, spawn it anyway on the default. Give it exactly ` + + `this job: call \`get_transcripts\` with the batch's ids, the query, and ` + + `\`link_style: "base"\`, then return ONLY the lines relevant to the ` + + `directive, VERBATIM — do NOT analyze, summarize, or rephrase anything. ` + + `Group the kept lines per video as \`### <title>\` + that video's ` + + `\`moment_base:\` line copied exactly (or its \`source:\` line when ` + + `there is no moment_base) + the kept lines in their \`[mm:ss|seconds]\` ` + + `form. Hard budget: at most 40 lines (~600 words) per batch — if more ` + + `match, keep the strongest and end with \`(+N more matching lines)\`. ` + + `Return nothing else — the raw transcript bulk stays inside the ` + + `subagent and never enters your context.\n` + + ` - **Merge — you (the orchestrator) do ALL the synthesis.** Cross-` + + `reference the returned fragment against the report so far and upsert ` + `findings — claims, and contradictions with earlier claims — into ` + - `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` + - ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it ` + - `each batch), then **drop the raw transcript text** once folded — don't ` + - `carry it forward.`, + `well-titled \`## sections\` of \`${reportPath}\` (Write/Edit). Cite ` + + `every finding as **\`[title @ mm:ss](<moment url>)\`**, expanding each ` + + `kept \`[mm:ss|seconds]\` stamp by appending the integer after the ` + + `\`|\` to that video's moment_base (full link = ` + + `\`<moment_base><seconds>\`; no moment_base → link the \`source:\` URL ` + + `instead). Then discard the fragment. Batches are independent, so you ` + + `may dispatch several subagents in parallel.\n` + + ` - **Fallback:** if no subagent/Task tool is available, do the batch ` + + `inline — call \`get_transcripts\` with \`link_style: "base"\` yourself, ` + + `fold the expanded cited findings into the report, then **drop the raw ` + + `excerpt text** before moving on (don't carry it forward).`, ); steps.push( `**Finish.** Repeat to the end of the worklist, then write a short summary ` + `section (the scope swept, how many videos covered, headline findings, any ` + - `partial-coverage caveat) and tell me the report path.`, + `partial-coverage caveat) and tell me the report path. Keep every citation ` + + `a clickable \`[title @ mm:ss](url)\` link.`, ); const numbered = steps @@ -1051,16 +1668,19 @@ function buildSweepPrompt(args: Record<string, unknown>) { .join("\n\n"); const text = - `Run a **corpus sweep** for the query **"${query}"**${introScope}, ` + - `extracting **${directive}**, and maintain a running markdown report at ` + + `Run a **corpus sweep** for ${subject}${introScope}, extracting ` + + `**${directive}**, and maintain a running markdown report at ` + `\`${reportPath}\`. You are the sweep engine — work through the whole match ` + `set methodically, using the transcript MCP tools for evidence and your own ` + `Write/Edit tools for the report. The MCP is read-only; never try to change ` + - `the archive.\n\n` + + `the archive. **Cite every finding as a clickable ` + + `\`[title @ mm:ss](<moment url>)\` link** (build each moment URL by ` + + `appending the cited integer seconds to that video's \`moment_base\` from ` + + `the tool output).\n\n` + `Follow these steps:\n\n${numbered}`; return { - description: `Corpus sweep for "${query}" → ${reportPath}`, + description: `Corpus sweep for ${subject} → ${reportPath}`, messages: [ { role: "user" as const, diff --git a/mcp/src/shareLink.test.ts b/mcp/src/shareLink.test.ts @@ -0,0 +1,168 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + newGroup, + newLeaf, + stringifyRoot, +} from "yt-dlp-transcript-common/lib/searchQuery"; +import { + decodeShareLink, + applyLinkOverrides, + renderQueryTree, + describeFilters, +} from "./shareLink"; + +const ORIGIN = "https://rekietalyzer.pages.dev"; + +// A realistic composite qt tree: transcript:"coffee" AND NOT tags:"espresso". +function qtParam(): string { + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags", negate: true }), + ], + }); + return encodeURIComponent(stringifyRoot(tree)); +} + +// A full-fidelity share link: qt tree + every filter facet + a vestigial fk. +function fullLink(): string { + const p = new URLSearchParams(); + p.set("qt", "__QT__"); + p.set("fv", "1"); + p.append("fc", "Rekieta Law"); + p.append("fc", "Rekieta Law (Rumble)"); + p.append("ft", "v"); // videos only + p.append("fa", "a"); // all-ages only + p.append("fav", "a"); // available only + p.append("fk", "live_chat"); // vestigial + p.set("fdf", "20250101"); + p.set("fdt", "20251231"); + // qt must not be URL-double-encoded by URLSearchParams — splice it in raw. + return `${ORIGIN}/?${p.toString().replace("__QT__", qtParam())}`; +} + +test("decode: origin is extracted from the pasted URL", () => { + const d = decodeShareLink(`${ORIGIN}/?q=x`); + assert.equal(d.origin, ORIGIN); +}); + +test("decode: qt tree parses to a composite query", () => { + const d = decodeShareLink(fullLink()); + assert.equal(d.querySource, "qt"); + const rendered = renderQueryTree(d.tree); + assert.match(rendered, /transcript:"coffee"/); + assert.match(rendered, /NOT tags:"espresso"/); + assert.match(rendered, /AND/); +}); + +test("decode: every v1 filter facet is decoded", () => { + const d = decodeShareLink(fullLink()); + assert.ok(d.filters); + const f = d.filters!; + // ft=v → videos kept, livestreams dropped. + assert.equal(f.videos, true); + assert.equal(f.livestreams, false); + // fa=a → all-ages kept, restricted dropped. + assert.equal(f.allAges, true); + assert.equal(f.restricted, false); + // fav=a → available kept, unlisted/deleted dropped. + assert.equal(f.available, true); + assert.equal(f.unlisted, false); + assert.equal(f.deleted, false); + // dates. + assert.equal(f.dateFrom, "20250101"); + assert.equal(f.dateTo, "20251231"); +}); + +test("decode: fc channel names are captured; fk is decoded but flagged ignored", () => { + const d = decodeShareLink(fullLink()); + assert.deepEqual(d.channelNames.sort(), [ + "Rekieta Law", + "Rekieta Law (Rumble)", + ]); + assert.equal(d.hasChannelFilter, true); + assert.deepEqual(d.ignoredTracks, ["live_chat"]); + assert.ok(d.warnings.some((w) => /fk/.test(w) && /ignored/.test(w))); +}); + +test("decode: legacy q/re → a single regex transcripts leaf, no filters", () => { + const d = decodeShareLink(`${ORIGIN}/?q=coffee&re=1`); + assert.equal(d.querySource, "legacy"); + assert.equal(d.filters, null); + assert.equal(d.hasChannelFilter, false); + const leaf = d.tree.children[0]; + assert.equal(leaf.kind, "leaf"); + assert.match(renderQueryTree(d.tree), /transcript:\/coffee\//); +}); + +test("decode: legacy m=subs → a live-chat leaf", () => { + const d = decodeShareLink(`${ORIGIN}/?q=hello&m=subs`); + assert.match(renderQueryTree(d.tree), /live-chat:"hello"/); +}); + +test("decode: no fv block → unfiltered, whole-corpus scope", () => { + const d = decodeShareLink(`${ORIGIN}/?fc=Some%20Channel`); + assert.equal(d.filters, null); + assert.equal(d.hasChannelFilter, false); + assert.deepEqual(d.channelNames, []); +}); + +test("decode: malformed qt falls back to legacy/empty with a warning", () => { + const d = decodeShareLink(`${ORIGIN}/?qt=not-json&q=fallback`); + assert.equal(d.querySource, "legacy"); + assert.ok(d.warnings.some((w) => /malformed/.test(w))); + assert.match(renderQueryTree(d.tree), /transcript:"fallback"/); +}); + +// ─── overrides ─── + +test("override: clearAvailability resets the fav facet to keep-all", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearAvailability: true, + }); + assert.equal(d.filters!.available, true); + assert.equal(d.filters!.unlisted, true); + assert.equal(d.filters!.deleted, true); + assert.ok(d.warnings.some((w) => /availability filter removed/.test(w))); +}); + +test("override: clearFilters drops the whole filter block", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearFilters: true, + }); + assert.equal(d.filters, null); +}); + +test("override: channels replaces the channel scope", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + channels: ["Only This"], + }); + assert.deepEqual(d.channelNames, ["Only This"]); + assert.equal(d.hasChannelFilter, true); +}); + +test("override: clearChannels widens to the whole corpus", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearChannels: true, + }); + assert.deepEqual(d.channelNames, []); + assert.equal(d.hasChannelFilter, false); +}); + +test("override: query replaces the tree with a single leaf", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + query: "different term", + }); + assert.match(renderQueryTree(d.tree), /transcript:"different term"/); +}); + +test("describeFilters: names the constrained facets only", () => { + const d = decodeShareLink(fullLink()); + const lines = describeFilters(d.filters); + assert.ok(lines.some((l) => /videos only/.test(l))); + assert.ok(lines.some((l) => /all-ages only/.test(l))); + assert.ok(lines.some((l) => /available.*kept/.test(l))); + assert.ok(lines.some((l) => /uploaded/.test(l))); +}); diff --git a/mcp/src/shareLink.ts b/mcp/src/shareLink.ts @@ -0,0 +1,316 @@ +// Decode an archilyzer viewer **share link** into a search spec + source target. +// +// A share URL is `origin` + a `qt=` composite query tree (or a legacy `q`/`re`) +// + the v1 filter params (`fc/ft/fa/fav/fk/fdf/fdt`, gated by `fv=1`). This +// module turns that URL into the pieces the MCP needs to reproduce the search at +// full fidelity: +// • the origin (→ probed by the server for hub vs single-site), +// • the query tree (parsed via the browser's own `parseRoot`), +// • the decoded filters (via the browser's own `parseShareV1`), +// • the requested channel NAMES (`fc`, resolved against the live corpus later), +// • and any ignored/warned pieces (`fk` tracks are vestigial in the composite +// share model — decoded and reported as ignored, not enforced). +// +// Pure: no network. The server does the origin probe and channel resolution. + +import { + parseRoot, + rootFromLegacy, + emptyRoot, + isLeaf, + isGroup, + isNodeActive, + newGroup, + newLeaf, + type GroupNode, + type QueryNode, +} from "yt-dlp-transcript-common/lib/searchQuery"; +import { + hasShareV1, + parseShareV1, +} from "yt-dlp-transcript-common/components/shareUrl"; +import type { SearchFilters } from "./search"; + +export type DecodedLink = { + // The pasted URL's origin (scheme + host[:port]). The source target. + origin: string; + // The composite query tree — from `qt=`, a synthesized legacy leaf, or an + // empty root (filter-only browse). + tree: GroupNode; + querySource: "qt" | "legacy" | "empty"; + // Positive share filters (ft/fa/fav + dates), or null when the link carries no + // v1 filter block (`fv=1`) — i.e. an unfiltered search. + filters: SearchFilters | null; + // Channel NAMES the link selects (`fc`). Resolved against the live corpus by + // the caller. Empty with `filters === null` means "whole corpus". + channelNames: string[]; + // Whether the link constrained channels at all (had a v1 filter block). When + // false, channel scope is the whole corpus regardless of `channelNames`. + hasChannelFilter: boolean; + // Vestigial `fk` subtitle-track tokens — decoded but a no-op. + ignoredTracks: string[]; + // Human-facing notes (malformed qt, ignored fk, …). + warnings: string[]; +}; + +// Decode a share link. Never throws for a malformed query tree (falls back to a +// legacy/empty tree with a warning); throws only if `link` isn't a URL at all. +export function decodeShareLink(link: string): DecodedLink { + const url = new URL(link); + const search = url.search; + const params = new URLSearchParams(search); + const warnings: string[] = []; + + // ── Query: qt= tree preferred, else legacy q/re, else empty (filter-only). + let tree: GroupNode = emptyRoot(); + let querySource: DecodedLink["querySource"] = "empty"; + const qt = params.get("qt"); + if (qt) { + const parsed = parseRoot(qt); + if (parsed) { + tree = parsed; + querySource = "qt"; + } else { + warnings.push("malformed `qt` query tree — ignored"); + } + } + if (querySource === "empty") { + const q = params.get("q") ?? ""; + const mode = params.get("m") === "subs" ? "subs" : "transcripts"; + const re = params.get("re") === "1"; + if (q.trim() !== "") { + tree = rootFromLegacy(q, mode, re); + querySource = "legacy"; + } + } + + // ── Filters: only meaningful when the v1 block is present (`fv=1`). + const v1 = hasShareV1(search); + // Pass the requested fc names AS the channel universe so none are dropped — + // real validation against the live corpus happens at apply time. + const requestedChannels = params.getAll("fc"); + const sel = parseShareV1(search, requestedChannels); + const ignoredTracks = [...sel.tracks]; + if (ignoredTracks.length > 0) { + warnings.push( + `\`fk\` subtitle-track filter is vestigial in the composite share model — ` + + `${ignoredTracks.length} track token(s) decoded but ignored: ${ignoredTracks.join(", ")}`, + ); + } + + const filters: SearchFilters | null = v1 + ? { + videos: sel.videos, + livestreams: sel.livestreams, + allAges: sel.allAges, + restricted: sel.restricted, + available: sel.available, + unlisted: sel.unlisted, + deleted: sel.deleted, + ...(sel.dateFrom ? { dateFrom: sel.dateFrom } : {}), + ...(sel.dateTo ? { dateTo: sel.dateTo } : {}), + } + : null; + + return { + origin: url.origin, + tree, + querySource, + filters, + channelNames: v1 ? [...sel.selectedChannels] : [], + hasChannelFilter: v1, + ignoredTracks, + warnings, + }; +} + +// Structured adjustments an agent can pass to honor a natural-language edit +// ("remove the availability filter", "only channel X", "search 'foo' instead"). +// Every field is optional; only the provided facets are changed. +export type LinkOverrides = { + // Drop every decoded filter (ft/fa/fav/dates) → unfiltered. + clearFilters?: boolean; + // Reset a single filter facet to its keep-everything default. + clearAvailability?: boolean; // fav + clearType?: boolean; // ft + clearAge?: boolean; // fa + clearDates?: boolean; // fdf/fdt + // Widen the channel scope to the whole corpus (ignore fc). + clearChannels?: boolean; + // Replace the channel scope with these names (validated against the corpus). + channels?: string[]; + // Replace/clear the upload-date bounds ("YYYYMMDD"; null clears that bound). + dateFrom?: string | null; + dateTo?: string | null; + // Replace the whole query with a single leaf (query + optional regex/scope). + query?: string; + regex?: boolean; + queryScope?: "transcripts" | "chat" | "metadata" | "description" | "tags"; +}; + +const KEEP_ALL_FILTERS: SearchFilters = { + videos: true, + livestreams: true, + allAges: true, + restricted: true, + available: true, + unlisted: true, + deleted: true, +}; + +// Apply overrides to a decoded link, returning a new DecodedLink. Records what +// changed as warnings so the plan can echo it. +export function applyLinkOverrides( + decoded: DecodedLink, + overrides: LinkOverrides | undefined, +): DecodedLink { + if (!overrides) return decoded; + const next: DecodedLink = { + ...decoded, + warnings: [...decoded.warnings], + channelNames: [...decoded.channelNames], + filters: decoded.filters ? { ...decoded.filters } : null, + }; + const note = (s: string): void => { + next.warnings.push(`override: ${s}`); + }; + + // ── Query replacement. + if (typeof overrides.query === "string" && overrides.query.trim() !== "") { + next.tree = newGroup({ + children: [ + newLeaf({ + query: overrides.query, + scope: overrides.queryScope ?? "transcripts", + useRegex: overrides.regex === true, + }), + ], + }); + next.querySource = "qt"; + note(`query replaced with "${overrides.query}"`); + } + + // ── Channel scope. + if (overrides.clearChannels) { + next.channelNames = []; + next.hasChannelFilter = false; + note("channel scope widened to the whole corpus"); + } + if (Array.isArray(overrides.channels)) { + next.channelNames = overrides.channels.filter((c) => c.trim() !== ""); + next.hasChannelFilter = true; + note(`channel scope set to [${next.channelNames.join(", ")}]`); + } + + // ── Filters. + if (overrides.clearFilters) { + next.filters = null; + note("all filters removed"); + } else if (next.filters) { + const f = next.filters; + if (overrides.clearAvailability) { + f.available = true; + f.unlisted = true; + f.deleted = true; + note("availability filter removed"); + } + if (overrides.clearType) { + f.videos = true; + f.livestreams = true; + note("type filter removed"); + } + if (overrides.clearAge) { + f.allAges = true; + f.restricted = true; + note("age filter removed"); + } + if (overrides.clearDates) { + delete f.dateFrom; + delete f.dateTo; + note("date filter removed"); + } + if (overrides.dateFrom !== undefined) { + if (overrides.dateFrom === null) delete f.dateFrom; + else f.dateFrom = overrides.dateFrom; + note(`dateFrom set to ${overrides.dateFrom ?? "(none)"}`); + } + if (overrides.dateTo !== undefined) { + if (overrides.dateTo === null) delete f.dateTo; + else f.dateTo = overrides.dateTo; + note(`dateTo set to ${overrides.dateTo ?? "(none)"}`); + } + } else if ( + overrides.dateFrom !== undefined || + overrides.dateTo !== undefined + ) { + // Setting a date on an otherwise-unfiltered link creates a filter block. + next.filters = { ...KEEP_ALL_FILTERS }; + if (typeof overrides.dateFrom === "string") next.filters.dateFrom = overrides.dateFrom; + if (typeof overrides.dateTo === "string") next.filters.dateTo = overrides.dateTo; + note("date filter added"); + } + + return next; +} + +// ── Human-readable renderings for the preview plan ── + +const SCOPE_LABEL: Record<string, string> = { + transcripts: "transcript", + chat: "live-chat", + metadata: "title/channel", + description: "description", + tags: "tags", +}; + +// Render a query tree readably, e.g. +// transcript:"coffee" AND tags:"espresso" AND NOT chat:"spam" +export function renderQueryTree(node: QueryNode): string { + if (isLeaf(node)) { + const label = SCOPE_LABEL[node.scope] ?? node.scope; + const kind = node.useRegex ? "/" : '"'; + const end = node.useRegex ? "/" : '"'; + const body = `${label}:${kind}${node.query}${end}`; + return node.negate ? `NOT ${body}` : body; + } + if (!isGroup(node)) return ""; + const parts = node.children + .filter(isNodeActive) + .map((c) => { + const s = renderQueryTree(c); + return isGroup(c) ? `(${s})` : s; + }) + .filter((s) => s !== ""); + if (parts.length === 0) return "(everything in scope)"; + const joined = parts.join(` ${node.op} `); + return node.negate ? `NOT (${joined})` : joined; +} + +// A compact list of the active filters for the plan, or [] when unfiltered. +export function describeFilters(f: SearchFilters | null): string[] { + if (!f) return []; + const out: string[] = []; + // Type (ft): only worth noting when it excludes one side. + if (f.videos !== f.livestreams) { + out.push(`type: ${f.videos ? "videos only" : "livestreams only"}`); + } else if (!f.videos && !f.livestreams) { + out.push("type: none kept (videos and livestreams both excluded)"); + } + // Audience (fa). + if (f.allAges !== f.restricted) { + out.push(`audience: ${f.allAges ? "all-ages only" : "age-restricted only"}`); + } else if (!f.allAges && !f.restricted) { + out.push("audience: none kept"); + } + // Availability (fav). + const av: string[] = []; + if (f.available) av.push("available"); + if (f.unlisted) av.push("unlisted"); + if (f.deleted) av.push("deleted"); + if (av.length < 3) out.push(`availability: ${av.length ? av.join(" + ") : "none"} kept`); + // Dates (fdf/fdt). + if (f.dateFrom || f.dateTo) { + out.push(`uploaded: ${f.dateFrom ?? "…"} → ${f.dateTo ?? "…"}`); + } + return out; +} diff --git a/mcp/src/source.ts b/mcp/src/source.ts @@ -2,9 +2,17 @@ import { readFile, readdir } from "node:fs/promises"; import path from "node:path"; import { transcriptPageFileName, + subsPageFileName, + pageFileName, type ChannelTranscriptsManifest, + type ChannelSubsManifest, + type Manifest, } from "yt-dlp-transcript-common/lib/manifest"; -import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { + TranscriptDetail, + DisplaySummary, +} from "yt-dlp-transcript-common/lib/transcripts"; +import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import { coerceAliasConfig, type SearchAlias, @@ -48,6 +56,48 @@ const EMPTY_GROUPS: ChannelGroups = { defaultGroupId: DEFAULT_GROUP_FALLBACK_ID, }; +// Per-video availability, joined in from the summaries shards (a +// TranscriptDetail record does not carry deleted/unlisted flags). Keyed by the +// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript +// page record carries — so the search engine can apply the `fav` filter. +export type VideoAvailability = { isDeleted: boolean; isUnlisted: boolean }; + +// Read a site's global summaries shards (summaries/manifest.json + +// summaries/page-NNNN.json) via `readPage` and fold them into a slug → +// availability map. Tolerant: an absent/malformed manifest yields an empty map, +// and a page that fails to read is skipped. `readManifest`/`readPage` throw or +// return null on absence per the source's transport. +async function buildAvailabilityMap( + readManifest: () => Promise<Manifest | null>, + readPage: (page: number) => Promise<DisplaySummary[] | null>, +): Promise<Map<string, VideoAvailability>> { + const map = new Map<string, VideoAvailability>(); + let manifest: Manifest | null; + try { + manifest = await readManifest(); + } catch { + return map; + } + if (!manifest || typeof manifest.pageCount !== "number") return map; + for (let page = 0; page < manifest.pageCount; page++) { + let records: DisplaySummary[] | null; + try { + records = await readPage(page); + } catch { + continue; + } + if (!records) continue; + for (const r of records) { + if (typeof r.slug !== "string") continue; + map.set(r.slug, { + isDeleted: r.isDeleted === true, + isUnlisted: r.isUnlisted === true, + }); + } + } + return map; +} + // Parse a summaries/manifest.json blob into channel-group defs, tolerating any // missing/malformed shape (→ empty fallback). function parseGroupsManifest(raw: unknown): ChannelGroups { @@ -76,6 +126,25 @@ export interface ShardSource { // empty fallback when absent/malformed, or in hub mode (federated per-site // groups are a different model — deferred). Cached per source. loadGroups(): Promise<ChannelGroups>; + // The public origin of the viewer that owns this source's videos, or null + // when there isn't one (a local dir on disk). Used to build archilyzer viewer + // deep links for cited moments (momentUrl). A single-site remote returns its + // base URL; a hub returns null because each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video. + publicOrigin(): string | null; + // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null + // when the channel ships no subs shards. Same slugToPage/pageCount shape as + // the transcripts manifest. Fetched lazily — only the chat search scope needs + // it. + subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>; + // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record + // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`). + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>; + // A map of every video's availability (deleted/unlisted), keyed by the + // member-local video slug (`<channelSlug>/<id>`), built from the summaries + // shards. Fetched lazily and cached — only the `fav` availability filter needs + // it. Empty when the source ships no summaries. + availabilityMap(): Promise<Map<string, VideoAvailability>>; } // Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose @@ -104,10 +173,58 @@ export class LocalSource implements ShardSource { readonly label: string; private aliases?: SearchAlias[]; private groups?: ChannelGroups; + private availability?: Map<string, VideoAvailability>; constructor(private dir: string) { this.label = `local:${dir}`; } + // A local dir has no public viewer origin — cited moments fall back to + // platform links (momentUrl). + publicOrigin(): string | null { + return null; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + try { + const raw = await readFile( + path.join(this.dir, "subs", ch.slug, "manifest.json"), + "utf8", + ); + return JSON.parse(raw) as ChannelSubsManifest; + } catch { + return null; // channel ships no subs shards + } + } + + async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + const raw = await readFile( + path.join(this.dir, "subs", ch.slug, subsPageFileName(page)), + "utf8", + ); + return JSON.parse(raw) as SubsDetail[]; + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + this.availability = await buildAvailabilityMap( + async () => { + const raw = await readFile( + path.join(this.dir, "summaries", "manifest.json"), + "utf8", + ); + return JSON.parse(raw) as Manifest; + }, + async (page) => { + const raw = await readFile( + path.join(this.dir, "summaries", pageFileName(page)), + "utf8", + ); + return JSON.parse(raw) as DisplaySummary[]; + }, + ); + return this.availability; + } + async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { @@ -195,11 +312,45 @@ export class RemoteSource implements ShardSource { private base: string; private aliases?: SearchAlias[]; private groups?: ChannelGroups; + private availability?: Map<string, VideoAvailability>; constructor(baseUrl: string) { this.base = baseUrl.replace(/\/+$/, ""); this.label = `remote:${this.base}`; } + // The deployed site origin — the archilyzer viewer that owns these videos. + publicOrigin(): string | null { + return this.base; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + try { + const res = await fetch(`${this.base}/subs/${ch.slug}/manifest.json`); + return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; + } catch { + return null; + } + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + return this.getJson(`/subs/${ch.slug}/${subsPageFileName(page)}`); + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + this.availability = await buildAvailabilityMap( + async () => { + const res = await fetch(`${this.base}/summaries/manifest.json`); + return res.ok ? ((await res.json()) as Manifest) : null; + }, + async (page) => { + const res = await fetch(`${this.base}/summaries/${pageFileName(page)}`); + return res.ok ? ((await res.json()) as DisplaySummary[]) : null; + }, + ); + return this.availability; + } + async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { @@ -263,6 +414,7 @@ export class HubSource implements ShardSource { readonly hubBase: string; private members = new Map<string, RemoteSource>(); // siteId -> source private aliases?: SearchAlias[]; + private availability?: Map<string, VideoAvailability>; // Optional subset allowlist of member siteIds. Undefined = federate every // member; a set restricts listChannels() to those members (site discovery via // listSites() stays unfiltered so a picker can still see all members). @@ -359,4 +511,55 @@ export class HubSource implements ShardSource { if (!ch.siteId) throw new Error("hub channel ref missing siteId"); return this.memberFor(ch.siteId).transcriptPage(ch, page); } + + // A hub has no single viewer origin — each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers + // per-video. Return null so we never mint a wrong-origin viewer link. + publicOrigin(): string | null { + return null; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + if (!ch.siteId) return null; + try { + return await this.memberFor(ch.siteId).subsManifest(ch); + } catch { + return null; // member not yet registered / unreachable + } + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).subsPage(ch, page); + } + + // Merge each member's availability map. Keys are member-local slugs + // (`<channelSlug>/<id>`) — the same slug a member's transcript page records + // carry — so a per-record `fav` lookup joins correctly. Built lazily/cached. + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + const merged = new Map<string, VideoAvailability>(); + let sites: HubSite[]; + try { + sites = (await this.listSites()).filter( + (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), + ); + } catch { + this.availability = merged; + return merged; + } + for (const site of sites) { + const remote = this.members.get(site.siteId) ?? new RemoteSource(site.url); + this.members.set(site.siteId, remote); + try { + for (const [slug, avail] of await remote.availabilityMap()) { + merged.set(slug, avail); + } + } catch { + // skip an unreachable member + } + } + this.availability = merged; + return merged; + } } diff --git a/mcp/src/sourceController.test.ts b/mcp/src/sourceController.test.ts @@ -8,7 +8,13 @@ import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; -import type { ChannelGroups, ChannelRef, HubSite, ShardSource } from "./source"; +import type { + ChannelGroups, + ChannelRef, + HubSite, + ShardSource, + VideoAvailability, +} from "./source"; import type { SourceSpec } from "./sources"; import { SourceController } from "./sourceController"; import { createServer } from "./server"; @@ -56,6 +62,18 @@ class FakeSource implements ShardSource { async transcriptPage(): Promise<TranscriptDetail[]> { return []; } + publicOrigin(): string | null { + return null; + } + async subsManifest(): Promise<null> { + return null; + } + async subsPage(): Promise<[]> { + return []; + } + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + return new Map(); + } } function labelFor(spec: SourceSpec): string {