Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit bdf9511f75e4531659bd3b1d61ee9dd77682ac43
parent 9c081346143f5ae9f8dc17c7d2f94cc03c13ab97
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  6 Jul 2026 21:34:18 -0400

Add /use-with-ai page + copy-for-AI buttons (player & search)

Layer 4 of bring-your-own-AI (client-side; export stays fully static):
- export/app/use-with-ai/page.tsx: static, hub-aware page explaining the chat,
  the llms.txt/corpus.json discovery files, and the MCP server (with an
  mcp.json snippet targeting this site/hub). Linked from Header + Footer.
- PlayerProvider.copyTranscriptMarkdown + a "Copy as Markdown" toolbar control
  in TranscriptModal β€” copies the shown transcript/live-chat via the shared
  transcriptToMarkdown formatter.
- TranscriptSearch "Copy for AI" button β€” copies matched videos + timestamped
  hit snippets as context.

Verified: common + export tsc clean; dev server renders /use-with-ai/ (200)
with the chat link, corpus.json links, and MCP snippet; home nav shows the link.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/PlayerProvider.tsx | 37+++++++++++++++++++++++++++++++++++++
Mcommon/components/TranscriptModal.tsx | 21+++++++++++++++++++++
Mcommon/components/TranscriptSearch.tsx | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 1+
Mexport/app/components/Footer.tsx | 9+++++++++
Mexport/app/components/Header.tsx | 6++++++
Aexport/app/use-with-ai/page.tsx | 139+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
7 files changed, 290 insertions(+), 0 deletions(-)

diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx @@ -18,6 +18,7 @@ import { useUrlParams, writeUrlParams } from "./urlState"; import type { ModalMode } from "./urlState"; import { formatDate, formatDuration } from "../lib/format"; import { cuesToSrt, cuesToText } from "../lib/vtt"; +import { transcriptToMarkdown } from "../lib/transcriptToMarkdown"; import type { Platform } from "../lib/transcripts"; import { vodExpiry } from "../lib/vodExpiry"; import { VodExpiredBadge } from "./badges"; @@ -129,6 +130,9 @@ type PlayerState = { // Download the currently-shown transcript or live-chat log (per modalMode) as // a single file, generated client-side from the loaded cues. downloadTranscriptFile: (fmt: "json" | "txt" | "srt") => void; + // Copy the currently-shown transcript (or live chat) to the clipboard as clean + // markdown for pasting into an AI chat. Resolves true on success. + copyTranscriptMarkdown: () => Promise<boolean>; openInPreservetube: () => void; seekTo: (seconds: number) => void; }; @@ -469,6 +473,37 @@ export function PlayerProvider({ [data, chat, activeSlug, modalMode], ); + // Copy the currently-shown cues as markdown (metadata header + timestamped + // body) for pasting into any AI chat. Uses the shared transcriptToMarkdown so + // the format matches the MCP server and the /ask chat. + const copyTranscriptMarkdown = useCallback(async (): Promise<boolean> => { + const isChat = modalMode === "chat"; + const cues = isChat + ? chat.slug === activeSlug + ? chat.cues + : null + : (data?.cues ?? null); + if (!data || !cues || cues.length === 0) return false; + const md = transcriptToMarkdown( + { + id: data.id, + title: isChat ? `${data.title} β€” live chat` : data.title, + channel: data.channel, + uploadDate: data.uploadDate, + webpageUrl: data.webpageUrl, + description: isChat ? undefined : data.description, + cues, + }, + { timestamps: true, includeDescription: !isChat }, + ); + try { + await navigator.clipboard.writeText(md); + return true; + } catch { + return false; + } + }, [data, chat, activeSlug, modalMode]); + const seekTo = useCallback((seconds: number) => { if (playerRef.current) { playerRef.current.seekTo(seconds, "seconds"); @@ -666,6 +701,7 @@ export function PlayerProvider({ copyDownloadCommand, copyShareUrl, downloadTranscriptFile, + copyTranscriptMarkdown, openInPreservetube, seekTo, }), @@ -689,6 +725,7 @@ export function PlayerProvider({ copyDownloadCommand, copyShareUrl, downloadTranscriptFile, + copyTranscriptMarkdown, openInPreservetube, seekTo, ], diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx @@ -42,6 +42,7 @@ export default function TranscriptModal() { copyDownloadCommand, copyShareUrl, downloadTranscriptFile, + copyTranscriptMarkdown, openInPreservetube, seekTo, } = usePlayer(); @@ -51,6 +52,8 @@ export default function TranscriptModal() { const copyResetRef = useRef<number | null>(null); const [shareCopied, setShareCopied] = useState(false); const shareResetRef = useRef<number | null>(null); + const [mdCopied, setMdCopied] = useState(false); + const mdResetRef = useRef<number | null>(null); const [downloadOpen, setDownloadOpen] = useState(false); const downloadRef = useRef<HTMLDivElement | null>(null); // Tracks whether the next scrollToIndex should animate. Smooth on @@ -283,6 +286,24 @@ export default function TranscriptModal() { </div> )} </div> + <ControlButton + title={ + !canDownloadFile + ? "Nothing to copy yet" + : mdCopied + ? "Copied!" + : `Copy this ${isChat ? "live chat" : "transcript"} as Markdown (for AI)` + } + onClick={async () => { + const ok = await copyTranscriptMarkdown(); + if (!ok) return; + setMdCopied(true); + if (mdResetRef.current) window.clearTimeout(mdResetRef.current); + mdResetRef.current = window.setTimeout(() => setMdCopied(false), 1500); + }} + char={mdCopied ? "βœ“" : "πŸ“‹"} + disabled={!canDownloadFile} + /> {data?.platform === "youtube" && ( <ControlButton title="Open in Preservetube" diff --git a/common/components/TranscriptSearch.tsx b/common/components/TranscriptSearch.tsx @@ -89,6 +89,51 @@ type ResultGroup = { accent?: string; }; +// Serialize the current result set as markdown to paste into an AI chat as +// context: the search terms, then each matching video with its timestamped hit +// snippets. Bounded so the clipboard payload stays reasonable. +function buildResultsContextMarkdown( + groups: ResultGroup[], + queries: string[], + opts: { maxVideos?: number; maxHitsPerVideo?: number } = {}, +): string { + const maxVideos = opts.maxVideos ?? 30; + const maxHits = opts.maxHitsPerVideo ?? 8; + const hms = (s: number): string => { + const n = Math.max(0, Math.floor(s)); + const h = Math.floor(n / 3600); + const m = Math.floor((n % 3600) / 60); + const ss = n % 60; + const mm = String(m).padStart(2, "0"); + const sss = String(ss).padStart(2, "0"); + return h > 0 ? `${h}:${mm}:${sss}` : `${m}:${sss}`; + }; + const lines: string[] = []; + lines.push( + queries.length + ? `# Transcript search results for: ${queries.join(" AND ")}` + : "# Transcript search results", + ); + lines.push(""); + const shown = groups.slice(0, maxVideos); + for (const g of shown) { + lines.push(`## ${g.title} β€” ${g.channel}${g.date ? ` (${g.date})` : ""}`); + const hits = g.hits.slice(0, maxHits); + for (const h of hits) { + const stamp = h.start > 0 ? `[${hms(h.start)}] ` : ""; + lines.push(`- ${stamp}${h.text.trim().replace(/\s+/g, " ")}`); + } + if (g.hits.length > hits.length) { + lines.push(`- …and ${g.hits.length - hits.length} more hit(s)`); + } + lines.push(""); + } + if (groups.length > shown.length) { + lines.push(`_(showing ${shown.length} of ${groups.length} matching videos)_`); + } + return lines.join("\n") + "\n"; +} + const DEFAULT_MAX_HITS = 500; const DEFAULT_FETCH_CONCURRENCY = 6; const DEFAULT_FLUSH_INTERVAL_MS = 120; @@ -1303,6 +1348,28 @@ export default function TranscriptSearch() { return m; }, [committedRoot]); + const [resultsCopied, setResultsCopied] = useState(false); + const resultsCopiedResetRef = useRef<number | null>(null); + const copyResultsContext = useCallback(async () => { + const queries = Array.from(leavesById.values()) + .map((l) => l.query.trim()) + .filter(Boolean); + const md = buildResultsContextMarkdown(resultGroups, queries); + try { + await navigator.clipboard.writeText(md); + setResultsCopied(true); + if (resultsCopiedResetRef.current) { + window.clearTimeout(resultsCopiedResetRef.current); + } + resultsCopiedResetRef.current = window.setTimeout( + () => setResultsCopied(false), + 1500, + ); + } catch { + /* clipboard blocked β€” no-op */ + } + }, [leavesById, resultGroups]); + // Hits split by leaf, in the order leaves appear in the tree, so the UI // sections render in a stable order regardless of the order findHitsInCues // discovered them. @@ -1780,6 +1847,16 @@ export default function TranscriptSearch() { Chart </button> </span> + {hasActiveQuery && resultGroups.length > 0 && ( + <button + type="button" + onClick={copyResultsContext} + title="Copy these results as context to paste into an AI chat" + className="inline-flex items-center rounded-md border border-border bg-muted px-2.5 py-1 text-xs font-normal text-muted-foreground hover:bg-accent hover:text-accent-foreground" + > + {resultsCopied ? "βœ“ Copied" : "Copy for AI"} + </button> + )} </h2> {view === "chart" ? ( diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -3,6 +3,7 @@ ## [Unreleased] - **Bring-your-own-AI: the archive is now machine-navigable for AI tools.** Every site publishes a small fixed set of discovery files β€” `llms.txt` (an LLM-readable overview) and `corpus.json` (a documented index of the channels and *how to fetch any transcript* from the existing paginated JSON shards), plus `robots.txt` and a `sitemap.xml`. Nothing is generated per video (the shard scheme is documented instead), so the file count stays constant no matter how large the corpus grows. This lets Claude Code and other tools browse and answer questions about the archive by fetching a couple of URLs. The federated hub publishes an aggregate `corpus.json`/`llms.txt` spanning every member site. - **New MCP server (`mcp/`) for Claude Code, Cursor, and other MCP clients.** A local tool that exposes the archive as MCP tools β€” `list_channels`, `search_transcripts` (timestamped snippets), `get_transcript`, `get_video_metadata` β€” reading the same static shards over disk or HTTP. It can point at one site or federate a whole hub. It never changes or hosts the site; see `mcp/README.md` for setup. +- **New "Use with AI" page + copy-for-AI buttons.** A `/use-with-ai` page (linked from the header and footer) explains the in-browser chat, the `llms.txt`/`corpus.json` discovery files, and the MCP server. The player toolbar gains a "Copy as Markdown" control that copies the current transcript or live chat as clean, timestamped markdown for pasting into any AI chat, and the search results header gains a "Copy for AI" button that copies the matched videos and their hit snippets as context. ## [0.6.4] - 2026-07-06 - **Fixed: sites always opened in light mode until you re-picked a theme.** If you'd chosen dark (or left it on "system" with a dark device), the page still loaded light on every visit and only switched after you opened the theme menu again. The saved theme is now re-applied before the page paints, so your choice sticks across reloads β€” no flash, no re-toggling. diff --git a/export/app/components/Footer.tsx b/export/app/components/Footer.tsx @@ -54,6 +54,15 @@ export default function Footer() { Offline </a> )} + <span aria-hidden="true" className="text-muted-foreground/50"> + Β· + </span> + <a + href="/use-with-ai" + className="underline underline-offset-2 hover:text-foreground transition-colors" + > + Use with AI + </a> </div> {socialLinks.length > 0 && ( <ul className="flex items-center gap-3 list-none"> diff --git a/export/app/components/Header.tsx b/export/app/components/Header.tsx @@ -68,6 +68,12 @@ export default function Header() { Downloads </Link> )} + <Link + href="/use-with-ai" + className="text-foreground hover:text-brand transition-colors" + > + Use with AI + </Link> </nav> <div className="ml-auto flex items-center gap-2 sm:gap-3 shrink-0"> diff --git a/export/app/use-with-ai/page.tsx b/export/app/use-with-ai/page.tsx @@ -0,0 +1,139 @@ +import type { Metadata } from "next"; +import Link from "next/link"; +import { currentSite } from "../lib/site"; +import { instanceMode } from "../lib/mode"; +import { hasArchives } from "../lib/archives"; + +export const metadata: Metadata = { title: "Use with AI" }; + +// A human-facing hub for the "bring your own AI" surface: the in-browser chat, +// the machine-readable discovery files (llms.txt / corpus.json), and the MCP +// server. Hub-aware β€” on a hub build it frames everything as federation-wide. +// Fully static server component; no data fetching. +export default function UseWithAiPage() { + const site = currentSite(); + const isHub = instanceMode() === "hub"; + const base = site.siteUrl?.replace(/\/+$/, "") ?? ""; + const abs = (p: string) => (base ? `${base}${p}` : p); + const scope = isHub ? "the whole federation" : "this archive"; + const envVar = isHub ? "TRANSCRIPT_HUB_URL" : "TRANSCRIPT_SITE_URL"; + const target = base || (isHub ? "https://your-hub.example" : "https://your-site.example"); + + const mcpSnippet = `{ + "mcpServers": { + "${site.siteId}": { + "command": "pnpm", + "args": [ + "-C", "/path/to/yt-dlp-transcript-browser", + "--filter", "yt-dlp-transcript-mcp", + "exec", "tsx", "src/index.ts" + ], + "env": { "${envVar}": "${target}" } + } + } +}`; + + return ( + <div className="mx-auto flex max-w-3xl flex-col gap-10"> + <header className="flex flex-col gap-3 border-b border-border pb-6"> + <p className="font-mono text-xs uppercase tracking-[0.18em] text-brand"> + Use with AI Β· {site.headerTitle} + </p> + <h1 className="font-display text-3xl font-semibold leading-tight text-foreground"> + Ask an AI about {scope} + </h1> + <p className="max-w-prose text-sm text-muted-foreground"> + Nothing is hosted or paid for here β€” you bring your own AI. Chat in your + browser with your own API key, or point a tool like Claude Code at the + machine-readable index and let it browse the transcripts itself. + </p> + </header> + + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + Chat in your browser + </h2> + <div className="flex flex-col gap-3 rounded-lg border border-border bg-card/40 p-5"> + <p className="text-sm text-muted-foreground"> + A retrieval-augmented chat that searches the transcripts and answers + with citations. It runs entirely in your browser and calls your own + provider (Anthropic, OpenAI, or Google Gemini) with a key you supply β€” + the key stays on your device and requests go straight to the provider. + </p> + <div> + <Link + href="/ask" + className="inline-flex items-center gap-2 rounded-md bg-primary px-4 py-2 text-sm font-medium text-primary-foreground transition-colors hover:bg-brand-strong" + > + Open the chat β†’ + </Link> + </div> + </div> + </section> + + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + For coding agents &amp; LLM tools + </h2> + <div className="flex flex-col gap-3 rounded-lg border border-border bg-card/40 p-5"> + <p className="text-sm text-muted-foreground"> + {scope[0].toUpperCase() + scope.slice(1)} publishes a small, fixed set + of discovery files. An agent (e.g. Claude Code via <code className="font-mono">WebFetch</code>) + can read these and navigate every transcript without any per-video + pages β€” the paginated shard scheme is documented inline. + </p> + <ul className="flex flex-col gap-2 text-sm"> + <li> + <a href={abs("/llms.txt")} className="font-mono text-brand hover:underline"> + /llms.txt + </a> + <span className="text-muted-foreground"> β€” an LLM-readable overview and link map.</span> + </li> + <li> + <a href={abs("/corpus.json")} className="font-mono text-brand hover:underline"> + /corpus.json + </a> + <span className="text-muted-foreground"> + {" "}β€” the channel index and exactly how to fetch any transcript + from the JSON shards{isHub ? " across every member site" : ""}. + </span> + </li> + </ul> + </div> + </section> + + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + MCP server + </h2> + <div className="flex flex-col gap-3 rounded-lg border border-border bg-card/40 p-5"> + <p className="text-sm text-muted-foreground"> + The repo ships an MCP server that exposes {scope} to Claude Code, + Claude Desktop, Cursor, and other MCP clients as tools + (<code className="font-mono">search_transcripts</code>, + {" "}<code className="font-mono">get_transcript</code>, …). It reads the + same static shards β€” over HTTP or from a local build + {isHub ? ", federating every member site" : ""}. Add it to a client: + </p> + <pre className="overflow-x-auto rounded-md border border-border bg-muted/50 p-3 font-mono text-xs text-foreground"> + <code>{mcpSnippet}</code> + </pre> + <p className="text-xs text-muted-foreground/80"> + Setup details and the <code className="font-mono">claude mcp add</code>{" "} + command are in <code className="font-mono">mcp/README.md</code>. + </p> + </div> + </section> + + {hasArchives() && ( + <p className="text-xs text-muted-foreground/70"> + Ingesting in bulk instead? The{" "} + <a href="/downloads" className="text-brand hover:underline"> + Downloads page + </a>{" "} + has whole-channel transcript zips. + </p> + )} + </div> + ); +}