commit 18e0266a1381373c8952403a498e2899102c0725
parent c8fd3ad224d0169633c909f15426334117122b8d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 6 Jul 2026 21:10:05 -0400
Add machine-readable AI-discovery surface (llms.txt/corpus.json) + shared transcript→markdown
Layer 0+1 of bring-your-own-AI:
- common/lib/transcriptToMarkdown.ts: one formatter (metadata header + optional
timestamps) reused by MCP/copy buttons/chat context. Unit-tested.
- common/lib/corpus.ts: pure builders/renderers for corpus.json (site + hub),
llms.txt, robots.txt, sitemap.xml — documents the existing paginated shard
scheme so nothing is emitted per-video (fixed file count). Unit-tested.
- compose-site.ts: emit the 4 site-root files after site.json + archives; add
CORS entries so a hub/remote MCP client can read corpus.json cross-origin.
- compose-hub.ts: emit an aggregate hub corpus.json/llms.txt directory over all
federated member origins (first-class hub PWA support).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 682 insertions(+), 1 deletion(-)
diff --git a/common/bin/compose-hub.ts b/common/bin/compose-hub.ts
@@ -15,7 +15,13 @@ import path from "node:path";
import { cp, rm, writeFile, access } from "node:fs/promises";
import { getPaths } from "../lib/paths";
import { listSites, resolveHubUrl } from "../lib/site";
+import { getHomepageConfig } from "../lib/homepage";
import { SITE_DESCRIPTOR_VERSION } from "../lib/siteDescriptor";
+import {
+ buildHubCorpus,
+ renderHubLlmsTxt,
+ renderRobotsTxt,
+} from "../lib/corpus";
// CORS for the hub's own JSON (hub-sites.json). The hub is primarily a reader,
// but keeping its endpoints CORS-open lets a hub-of-hubs federate it too. Same
@@ -33,6 +39,12 @@ const CORS_HEADERS = `# Generated by compose-hub.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/stats/*
Access-Control-Allow-Origin: *
+/corpus.json
+ Access-Control-Allow-Origin: *
+/llms.txt
+ Access-Control-Allow-Origin: *
+/robots.txt
+ Access-Control-Allow-Origin: *
`;
// One entry the hub registry loads at boot to seed its trusted built-in pool.
@@ -80,6 +92,28 @@ async function main(): Promise<void> {
JSON.stringify(builtins),
);
+ // Aggregate AI-discovery surface for the federation (fixed file count): a hub
+ // corpus.json directory of member origins + llms.txt + robots.txt. The hub
+ // holds no shard data — these point at each member's own corpus.json.
+ const hub = getHomepageConfig(paths);
+ const hubCorpus = buildHubCorpus(builtins, {
+ hubTitle: hub.siteTitle,
+ hubUrl: hub.siteUrl,
+ generatedAt: new Date().toISOString(),
+ });
+ await writeFile(
+ path.join(publicDir, "corpus.json"),
+ JSON.stringify(hubCorpus),
+ );
+ await writeFile(
+ path.join(publicDir, "llms.txt"),
+ renderHubLlmsTxt(hubCorpus),
+ );
+ await writeFile(
+ path.join(publicDir, "robots.txt"),
+ renderRobotsTxt({ siteUrl: hub.siteUrl }),
+ );
+
await writeFile(path.join(publicDir, "_headers"), CORS_HEADERS);
// The hub always ships a PWA. Copy the hub service worker into place. Until
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -24,7 +24,13 @@ import {
type DuplicateReport,
} from "../lib/duplicates";
import type { Manifest, SubsManifest } from "../lib/manifest";
-import { buildSiteDescriptor } from "../lib/siteDescriptor";
+import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescriptor";
+import {
+ buildSiteCorpus,
+ renderSiteLlmsTxt,
+ renderRobotsTxt,
+ renderSitemapXml,
+} from "../lib/corpus";
import { archiveTranscripts } from "../controller/archiveTranscripts";
import { archiveLiveChat } from "../controller/archiveLiveChat";
import {
@@ -53,6 +59,14 @@ const CORS_HEADERS = `# Generated by compose-site.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/archives/*
Access-Control-Allow-Origin: *
+/corpus.json
+ Access-Control-Allow-Origin: *
+/llms.txt
+ Access-Control-Allow-Origin: *
+/robots.txt
+ Access-Control-Allow-Origin: *
+/sitemap.xml
+ Access-Control-Allow-Origin: *
`;
// Emit the public federation contract: /site.json (branding + channels +
@@ -77,6 +91,68 @@ async function emitFederationFiles(
);
}
+// Emit the AI-discovery surface — a small FIXED set of site-root files
+// (llms.txt, corpus.json, robots.txt, sitemap.xml). These document how to
+// navigate the already-served paginated shards; they never enumerate per-video
+// files, so the count is constant regardless of corpus size. Runs after
+// site.json and the archives are composed (both feed into these files).
+async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
+ const sitePath = path.join(paths.exportPublicDir, "site.json");
+ if (!(await exists(sitePath))) return; // no composed data → nothing to describe
+ const descriptor = JSON.parse(
+ await readFile(sitePath, "utf8"),
+ ) as PublicSiteDescriptor;
+
+ // Bulk archives present? (mirror of export/app/lib/archives.ts hasArchives)
+ let hasArchives = false;
+ const amPath = path.join(
+ paths.exportPublicDir,
+ "archives",
+ ARCHIVE_MANIFEST_FILENAME,
+ );
+ if (await exists(amPath)) {
+ try {
+ const am = JSON.parse(await readFile(amPath, "utf8")) as ArchiveManifest;
+ hasArchives =
+ Array.isArray(am.entries) &&
+ am.entries.some((e) => !e.oversize || e.url);
+ } catch {
+ // A malformed archive manifest just means we omit the archives link.
+ }
+ }
+
+ const corpus = buildSiteCorpus(descriptor, { hasArchives });
+ await writeFile(
+ path.join(paths.exportPublicDir, "corpus.json"),
+ JSON.stringify(corpus),
+ );
+ await writeFile(
+ path.join(paths.exportPublicDir, "llms.txt"),
+ renderSiteLlmsTxt(corpus),
+ );
+ await writeFile(
+ path.join(paths.exportPublicDir, "robots.txt"),
+ renderRobotsTxt({ siteUrl: descriptor.siteUrl }),
+ );
+
+ // A sitemap of relative paths is useless, so only emit one with an absolute
+ // siteUrl; otherwise clear any stale copy from a previous build.
+ const sitemapPath = path.join(paths.exportPublicDir, "sitemap.xml");
+ if (descriptor.siteUrl) {
+ const routes = ["/", "/use-with-ai", "/changelog"];
+ if (hasArchives) routes.push("/downloads");
+ if (await exists(path.join(paths.exportPublicDir, DUPLICATES_FILENAME))) {
+ routes.push("/duplicates");
+ }
+ await writeFile(
+ sitemapPath,
+ renderSitemapXml({ siteUrl: descriptor.siteUrl, routes }),
+ );
+ } else {
+ await rm(sitemapPath, { force: true });
+ }
+}
+
// Whether this build ships an installable PWA. Resolved from the site's `pwa`
// config flag (site builds are dumb instances by default) or forced on in hub
// mode. Keep in sync with export/app/lib/mode.ts shipsPwa().
@@ -435,6 +511,10 @@ async function main(): Promise<void> {
// --- bulk-download archive zips (on by default; see archivesEnabled) ---
await composeArchives(site, memberSlugs, paths);
+ // --- AI discovery: llms.txt / corpus.json / robots.txt / sitemap.xml ---
+ // (after site.json + archives — both feed into these fixed-count files)
+ await emitAiFiles(paths);
+
const channelDirs = (await readdir(paths.exportTranscriptsDir).catch(
() => [] as string[],
)).length;
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -0,0 +1,115 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ buildSiteCorpus,
+ buildHubCorpus,
+ renderSiteLlmsTxt,
+ renderHubLlmsTxt,
+ renderRobotsTxt,
+ renderSitemapXml,
+ CORPUS_SPEC_VERSION,
+} from "./corpus";
+import type { PublicSiteDescriptor } from "./siteDescriptor";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/corpus.test.ts
+
+function descriptor(
+ over: Partial<PublicSiteDescriptor> = {},
+): PublicSiteDescriptor {
+ return {
+ contract: 1,
+ siteId: "demo",
+ siteTitle: "Demo Archive",
+ siteDescription: "Talks and streams.",
+ headerTitle: "Demo",
+ homeTagline: "",
+ pwa: false,
+ socialLinks: [],
+ groups: [],
+ defaultGroupId: "default",
+ channels: [
+ { slug: "alice", name: "Alice", count: 12 },
+ { slug: "bob", name: "Bob", count: 3, groupId: "g1" },
+ ],
+ generatedAt: "2026-07-06T00:00:00.000Z",
+ summariesVersion: 3,
+ ...over,
+ };
+}
+
+test("buildSiteCorpus: absolute shard URLs + totals when siteUrl is set", () => {
+ const corpus = buildSiteCorpus(
+ descriptor({ siteUrl: "https://demo.example/", hubUrl: "https://hub.example" }),
+ { hasArchives: true },
+ );
+ assert.equal(corpus.spec, CORPUS_SPEC_VERSION);
+ assert.equal(corpus.kind, "site");
+ assert.equal(corpus.totals.channels, 2);
+ assert.equal(corpus.totals.videos, 15);
+ assert.equal(corpus.site.url, "https://demo.example/");
+ assert.equal(corpus.site.hubUrl, "https://hub.example");
+ assert.equal(
+ corpus.channels[0].manifests.transcripts,
+ "https://demo.example/transcripts/alice/manifest.json",
+ );
+ assert.equal(corpus.channels[1].groupId, "g1");
+ assert.ok(corpus.bulkArchives, "archives present → bulkArchives block");
+ assert.ok(corpus.shardScheme.description.includes("paginated"));
+});
+
+test("buildSiteCorpus: root-relative URLs + no archives when unset", () => {
+ const corpus = buildSiteCorpus(descriptor(), { hasArchives: false });
+ assert.equal(corpus.site.url, undefined);
+ assert.equal(
+ corpus.channels[0].manifests.subs,
+ "/subs/alice/manifest.json",
+ );
+ assert.equal(corpus.bulkArchives, undefined);
+});
+
+test("renderSiteLlmsTxt: title, corpus link, channels", () => {
+ const corpus = buildSiteCorpus(
+ descriptor({ siteUrl: "https://demo.example" }),
+ { hasArchives: true },
+ );
+ const txt = renderSiteLlmsTxt(corpus);
+ assert.match(txt, /^# Demo Archive/);
+ assert.match(txt, /\[corpus\.json\]\(https:\/\/demo\.example\/corpus\.json\)/);
+ assert.match(txt, /Alice — 12 transcripts/);
+ assert.match(txt, /Bulk archives/);
+});
+
+test("buildHubCorpus: drops members with no siteUrl, links each corpus", () => {
+ const hub = buildHubCorpus(
+ [
+ { siteId: "a", siteTitle: "A", siteUrl: "https://a.example/", pwa: true },
+ { siteId: "b", siteTitle: "B", siteUrl: "", pwa: false },
+ ],
+ { hubTitle: "The Hub", hubUrl: "https://hub.example", generatedAt: "t" },
+ );
+ assert.equal(hub.kind, "hub");
+ assert.equal(hub.sites.length, 1);
+ assert.equal(hub.sites[0].corpus, "https://a.example/corpus.json");
+ assert.equal(hub.sites[0].siteJson, "https://a.example/site.json");
+ const txt = renderHubLlmsTxt(hub);
+ assert.match(txt, /^# The Hub/);
+ assert.match(txt, /A: https:\/\/a\.example/);
+});
+
+test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => {
+ assert.match(
+ renderRobotsTxt({ siteUrl: "https://demo.example/" }),
+ /Sitemap: https:\/\/demo\.example\/sitemap\.xml/,
+ );
+ assert.doesNotMatch(renderRobotsTxt({}), /Sitemap:/);
+});
+
+test("renderSitemapXml: one loc per route", () => {
+ const xml = renderSitemapXml({
+ siteUrl: "https://demo.example",
+ routes: ["/", "/use-with-ai"],
+ });
+ assert.match(xml, /<loc>https:\/\/demo\.example\/<\/loc>/);
+ assert.match(xml, /<loc>https:\/\/demo\.example\/use-with-ai<\/loc>/);
+});
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -0,0 +1,291 @@
+import type { PublicSiteDescriptor } from "./siteDescriptor";
+
+// The machine-readable corpus index emitted at `/corpus.json` on every export
+// bundle (and an aggregate variant on a hub). It does NOT contain transcripts —
+// the corpus is far too large to enumerate per-video without blowing the
+// Cloudflare Pages file-count limit. Instead it *documents how to navigate the
+// existing paginated JSON shards*, turning the already-served
+// `subs/`, `transcripts/`, and `summaries/` trees into a self-describing API for
+// LLM tools (Claude Code `WebFetch`, the repo MCP server, the in-browser chat).
+//
+// Pure module: builders take already-loaded data and return plain objects /
+// strings. All file I/O lives in compose-site.ts / compose-hub.ts.
+
+export const CORPUS_SPEC_VERSION = 1;
+
+// How to resolve a single transcript from the paginated shards, described once
+// and embedded in every corpus.json so any HTTP client can navigate without
+// reading our source.
+const SHARD_SCHEME = {
+ description:
+ "Transcripts are served as paginated JSON shards — there is no per-video " +
+ "file. To read one video's transcript: (1) GET the channel's transcripts " +
+ "manifest; (2) look up the video id in its `slugToPage` map to get a page " +
+ "number N; (3) GET page-<NNNN>.json (N zero-padded to 4 digits) and take " +
+ "the record whose `id` matches.",
+ transcriptsManifest:
+ "<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }",
+ transcriptPage:
+ "/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, " +
+ "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }",
+ subsManifest:
+ "<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme",
+ summariesIndex:
+ "/summaries/manifest.json + /summaries/page-<NNNN>.json -> cross-channel browse index",
+ pageNumberFormat: "zero-padded to 4 digits, e.g. page 0 -> page-0000.json",
+} as const;
+
+export type CorpusChannel = {
+ slug: string;
+ name: string;
+ videoCount: number;
+ groupId?: string;
+ manifests: {
+ transcripts: string;
+ subs: string;
+ };
+};
+
+export type SiteCorpus = {
+ spec: number;
+ kind: "site";
+ generatedAt: string;
+ site: {
+ id: string;
+ title: string;
+ description: string;
+ url?: string;
+ hubUrl?: string;
+ };
+ totals: { channels: number; videos: number };
+ channels: CorpusChannel[];
+ shardScheme: typeof SHARD_SCHEME;
+ // Present when this build ships bulk-download archives (whole-channel zips).
+ bulkArchives?: { manifest: string; note: string };
+ // Pointer to the human page and BYO-key chat.
+ useWithAi: string;
+};
+
+export type HubCorpusSite = {
+ siteId: string;
+ title: string;
+ url: string;
+ corpus: string;
+ siteJson: string;
+ pwa: boolean;
+};
+
+export type HubCorpus = {
+ spec: number;
+ kind: "hub";
+ generatedAt: string;
+ hub: { title: string; url?: string };
+ sites: HubCorpusSite[];
+ federation: { description: string };
+ useWithAi: string;
+};
+
+// Minimal member shape a hub knows about (mirror of compose-hub.ts HubSiteEntry).
+export type HubMemberInput = {
+ siteId: string;
+ siteTitle: string;
+ siteUrl: string;
+ pwa?: boolean;
+};
+
+// Join an origin base with a root-relative path. When no base is known (a site
+// built without a configured siteUrl) the path is left root-relative — still
+// correct for a same-origin fetch, just not portable cross-origin.
+function join(base: string | undefined, p: string): string {
+ if (!base) return p;
+ return `${base.replace(/\/+$/, "")}${p}`;
+}
+
+// Build the per-site corpus index from the public site descriptor (`/site.json`)
+// plus whether this build emitted bulk archives.
+export function buildSiteCorpus(
+ descriptor: PublicSiteDescriptor,
+ opts: { hasArchives: boolean },
+): SiteCorpus {
+ const base = descriptor.siteUrl;
+ const channels: CorpusChannel[] = descriptor.channels.map((c) => ({
+ slug: c.slug,
+ name: c.name,
+ videoCount: c.count,
+ ...(c.groupId ? { groupId: c.groupId } : {}),
+ manifests: {
+ transcripts: join(base, `/transcripts/${c.slug}/manifest.json`),
+ subs: join(base, `/subs/${c.slug}/manifest.json`),
+ },
+ }));
+ const videos = channels.reduce((n, c) => n + c.videoCount, 0);
+
+ const corpus: SiteCorpus = {
+ spec: CORPUS_SPEC_VERSION,
+ kind: "site",
+ generatedAt: descriptor.generatedAt,
+ site: {
+ id: descriptor.siteId,
+ title: descriptor.siteTitle,
+ description: descriptor.siteDescription,
+ ...(descriptor.siteUrl ? { url: descriptor.siteUrl } : {}),
+ ...(descriptor.hubUrl ? { hubUrl: descriptor.hubUrl } : {}),
+ },
+ totals: { channels: channels.length, videos },
+ channels,
+ shardScheme: SHARD_SCHEME,
+ useWithAi: join(base, "/use-with-ai"),
+ };
+ if (opts.hasArchives) {
+ corpus.bulkArchives = {
+ manifest: join(base, "/archives/manifest.json"),
+ note: "Whole-channel transcript and live-chat archives for offline bulk ingestion.",
+ };
+ }
+ return corpus;
+}
+
+// Build the aggregate hub corpus: a directory of member origins, each pointing
+// at its own corpus.json. No shard data — the hub reads members cross-origin.
+export function buildHubCorpus(
+ members: HubMemberInput[],
+ opts: { hubTitle: string; hubUrl?: string; generatedAt: string },
+): HubCorpus {
+ const sites: HubCorpusSite[] = members
+ .filter((m) => m.siteUrl)
+ .map((m) => {
+ const url = m.siteUrl.replace(/\/+$/, "");
+ return {
+ siteId: m.siteId,
+ title: m.siteTitle,
+ url,
+ corpus: `${url}/corpus.json`,
+ siteJson: `${url}/site.json`,
+ pwa: m.pwa === true,
+ };
+ });
+ return {
+ spec: CORPUS_SPEC_VERSION,
+ kind: "hub",
+ generatedAt: opts.generatedAt,
+ hub: { title: opts.hubTitle, ...(opts.hubUrl ? { url: opts.hubUrl } : {}) },
+ sites,
+ federation: {
+ description:
+ "This is a federation hub. Each site below is an independent origin " +
+ "serving its own corpus.json and CORS-enabled JSON shards. To search " +
+ "the whole federation, fetch each site's corpus.json and follow its " +
+ "shardScheme; results can be merged client-side.",
+ },
+ useWithAi: join(opts.hubUrl, "/use-with-ai"),
+ };
+}
+
+// Cap on how many channels/sites we inline into llms.txt. The full set is always
+// in corpus.json; llms.txt is a human/LLM-readable overview, not an exhaustive
+// index. (Bounded either way — this is per-channel, never per-video.)
+const LLMS_INLINE_LIMIT = 100;
+
+// Render the per-site llms.txt (llmstxt.org convention: H1 + blockquote summary
+// + linked sections).
+export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
+ const base = corpus.site.url;
+ const out: string[] = [];
+ out.push(`# ${corpus.site.title}`);
+ out.push("");
+ const summary =
+ (corpus.site.description ? corpus.site.description.trim() + " " : "") +
+ `A machine-navigable archive of ${corpus.totals.videos.toLocaleString()} ` +
+ `transcripts across ${corpus.totals.channels} channel(s). ` +
+ `See corpus.json for the shard API.`;
+ out.push(`> ${summary}`);
+ out.push("");
+ out.push("## Ask AI");
+ out.push(
+ `- [Use with AI](${join(base, "/use-with-ai")}): in-browser chat (bring ` +
+ `your own API key) and MCP-server setup for Claude Code, Cursor, and other tools.`,
+ );
+ out.push("");
+ out.push("## Corpus");
+ out.push(
+ `- [corpus.json](${join(base, "/corpus.json")}): machine-readable index — ` +
+ `channels and how to fetch any transcript from the paginated JSON shards.`,
+ );
+ if (corpus.bulkArchives) {
+ out.push(
+ `- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` +
+ `and live-chat zips for offline ingestion.`,
+ );
+ }
+ out.push("");
+ out.push("## Channels");
+ const shown = corpus.channels.slice(0, LLMS_INLINE_LIMIT);
+ for (const c of shown) {
+ out.push(`- ${c.name} — ${c.videoCount} transcripts: ${c.manifests.transcripts}`);
+ }
+ if (corpus.channels.length > shown.length) {
+ out.push(
+ `- …and ${corpus.channels.length - shown.length} more — see corpus.json for the full list.`,
+ );
+ }
+ return out.join("\n") + "\n";
+}
+
+// Render the aggregate hub llms.txt.
+export function renderHubLlmsTxt(corpus: HubCorpus): string {
+ const base = corpus.hub.url;
+ const out: string[] = [];
+ out.push(`# ${corpus.hub.title}`);
+ out.push("");
+ out.push(
+ `> A federated hub across ${corpus.sites.length} transcript archive(s). ` +
+ `Each site is an independent origin with its own machine-readable corpus. ` +
+ `See corpus.json for the federation directory.`,
+ );
+ out.push("");
+ out.push("## Ask AI");
+ out.push(
+ `- [Use with AI](${join(base, "/use-with-ai")}): ask across the whole ` +
+ `federation (bring your own API key) or wire up the MCP server.`,
+ );
+ out.push("");
+ out.push("## Federation");
+ out.push(
+ `- [corpus.json](${join(base, "/corpus.json")}): machine-readable directory ` +
+ `of every member site and its corpus endpoint.`,
+ );
+ out.push("");
+ out.push("## Sites");
+ for (const s of corpus.sites.slice(0, LLMS_INLINE_LIMIT)) {
+ out.push(`- ${s.title}: ${s.url} (corpus: ${s.corpus})`);
+ }
+ return out.join("\n") + "\n";
+}
+
+// robots.txt — allow crawling and advertise the LLM index + sitemap.
+export function renderRobotsTxt(opts: { siteUrl?: string }): string {
+ const out: string[] = ["User-agent: *", "Allow: /", ""];
+ out.push("# AI / LLM corpus index: /llms.txt and /corpus.json");
+ if (opts.siteUrl) {
+ out.push(`Sitemap: ${opts.siteUrl.replace(/\/+$/, "")}/sitemap.xml`);
+ }
+ return out.join("\n") + "\n";
+}
+
+// sitemap.xml of the app's static routes. Only meaningful with an absolute
+// siteUrl; callers skip emission otherwise. One file regardless of corpus size.
+export function renderSitemapXml(opts: {
+ siteUrl: string;
+ routes: string[];
+}): string {
+ const base = opts.siteUrl.replace(/\/+$/, "");
+ const urls = opts.routes
+ .map((r) => ` <url><loc>${base}${r === "/" ? "/" : r}</loc></url>`)
+ .join("\n");
+ return (
+ `<?xml version="1.0" encoding="UTF-8"?>\n` +
+ `<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n` +
+ `${urls}\n` +
+ `</urlset>\n`
+ );
+}
diff --git a/common/lib/transcriptToMarkdown.test.ts b/common/lib/transcriptToMarkdown.test.ts
@@ -0,0 +1,50 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { transcriptToMarkdown } from "./transcriptToMarkdown";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/transcriptToMarkdown.test.ts
+
+const base = {
+ id: "abc123",
+ title: "Episode One",
+ channel: "Alice",
+ uploadDate: "20191218",
+ duration: 3661,
+ webpageUrl: "https://youtube.com/watch?v=abc123",
+ description: "A first episode.",
+ cues: [
+ { start: 0, end: 2, text: "Hello there" },
+ { start: 3661.25, end: 3663, text: "General Kenobi" },
+ ],
+};
+
+test("renders header, description, and timestamped cues by default", () => {
+ const md = transcriptToMarkdown(base);
+ assert.match(md, /^# Episode One/);
+ assert.match(md, /- Channel: Alice/);
+ assert.match(md, /- Uploaded: 2019-12-18/);
+ assert.match(md, /- Duration: 1:01:01/);
+ assert.match(md, /- Source: https:\/\/youtube\.com/);
+ assert.match(md, /## Description\n\nA first episode\./);
+ assert.match(md, /\[0:00\] Hello there/);
+ assert.match(md, /\[1:01:01\] General Kenobi/);
+});
+
+test("timestamps:false emits plain prose lines", () => {
+ const md = transcriptToMarkdown(base, { timestamps: false });
+ assert.match(md, /\nHello there\n/);
+ assert.doesNotMatch(md, /\[0:00\]/);
+});
+
+test("missing cues → explicit no-transcript marker", () => {
+ const md = transcriptToMarkdown({ id: "x", title: "Empty", cues: undefined });
+ assert.match(md, /_\(no transcript available\)_/);
+});
+
+test("maxCues truncates and notes it", () => {
+ const md = transcriptToMarkdown(base, { maxCues: 1 });
+ assert.match(md, /Hello there/);
+ assert.doesNotMatch(md, /General Kenobi/);
+ assert.match(md, /truncated: showing 1 of 2 cues/);
+});
diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts
@@ -0,0 +1,108 @@
+import type { Cue } from "./vtt";
+import { formatDate, formatDuration } from "./format";
+
+// The subset of a transcripts-page record (TranscriptDetail, see transcripts.ts)
+// needed to render a self-contained markdown document. Kept as its own loose
+// type so callers on the client (which reconstruct records from shard JSON) and
+// on the server (MCP, build tools) can all feed it without importing the full
+// TranscriptDetail chain.
+export type TranscriptMarkdownInput = {
+ id: string;
+ title: string;
+ channel?: string;
+ channelSlug?: string;
+ uploadDate?: string; // "YYYYMMDD"
+ duration?: number; // seconds
+ webpageUrl?: string;
+ description?: string;
+ tags?: string[];
+ cues?: Cue[] | undefined;
+};
+
+export type TranscriptMarkdownOptions = {
+ // Prefix each cue line with a [h:mm:ss] timestamp. Default true — timestamps
+ // let an LLM cite a moment and let a reader jump to it. Turn off for the
+ // cleanest possible prose block.
+ timestamps?: boolean;
+ // Include the video description section. Default true.
+ includeDescription?: boolean;
+ // Include a "Tags" line. Default false — tags are noisy for most Q&A.
+ includeTags?: boolean;
+ // Cap the number of cue lines emitted (for fitting a context window). When
+ // truncated, a marker line is appended. Default: no cap.
+ maxCues?: number;
+};
+
+// [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so
+// handle the zero case explicitly here (a transcript's first cue is often 0s).
+function stamp(totalSeconds: number): string {
+ const s = Math.max(0, Math.floor(totalSeconds));
+ return s === 0 ? "0:00" : formatDuration(s);
+}
+
+// Render a transcript record as a clean, self-contained markdown document:
+// a metadata header, an optional description, and the transcript body. This is
+// the single source of truth for "transcript → text for an AI" across the MCP
+// server, the in-browser copy buttons, and the chat retrieval context.
+export function transcriptToMarkdown(
+ input: TranscriptMarkdownInput,
+ options: TranscriptMarkdownOptions = {},
+): string {
+ const {
+ timestamps = true,
+ includeDescription = true,
+ includeTags = false,
+ maxCues,
+ } = options;
+
+ const lines: string[] = [];
+ lines.push(`# ${input.title || input.id}`);
+ lines.push("");
+
+ const meta: string[] = [];
+ if (input.channel) meta.push(`- Channel: ${input.channel}`);
+ if (input.uploadDate) meta.push(`- Uploaded: ${formatDate(input.uploadDate)}`);
+ if (typeof input.duration === "number" && input.duration > 0) {
+ meta.push(`- Duration: ${formatDuration(input.duration)}`);
+ }
+ meta.push(`- Video ID: ${input.id}`);
+ if (input.webpageUrl) meta.push(`- Source: ${input.webpageUrl}`);
+ if (includeTags && input.tags && input.tags.length > 0) {
+ meta.push(`- Tags: ${input.tags.join(", ")}`);
+ }
+ lines.push(...meta);
+
+ if (includeDescription && input.description && input.description.trim()) {
+ lines.push("");
+ lines.push("## Description");
+ lines.push("");
+ lines.push(input.description.trim());
+ }
+
+ lines.push("");
+ lines.push("## Transcript");
+ lines.push("");
+
+ const cues = input.cues;
+ if (!cues || cues.length === 0) {
+ lines.push("_(no transcript available)_");
+ return lines.join("\n") + "\n";
+ }
+
+ const limit =
+ typeof maxCues === "number" && maxCues >= 0
+ ? Math.min(maxCues, cues.length)
+ : cues.length;
+ for (let i = 0; i < limit; i++) {
+ const cue = cues[i];
+ const text = cue.text.trim();
+ if (!text) continue;
+ lines.push(timestamps ? `[${stamp(cue.start)}] ${text}` : text);
+ }
+ if (limit < cues.length) {
+ lines.push("");
+ lines.push(`_(transcript truncated: showing ${limit} of ${cues.length} cues)_`);
+ }
+
+ return lines.join("\n") + "\n";
+}
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,8 @@
# Changelog
+## [Unreleased]
+- **Bring-your-own-AI: the archive is now machine-navigable for AI tools.** Every site publishes a small fixed set of discovery files — `llms.txt` (an LLM-readable overview) and `corpus.json` (a documented index of the channels and *how to fetch any transcript* from the existing paginated JSON shards), plus `robots.txt` and a `sitemap.xml`. Nothing is generated per video (the shard scheme is documented instead), so the file count stays constant no matter how large the corpus grows. This lets Claude Code and other tools browse and answer questions about the archive by fetching a couple of URLs. The federated hub publishes an aggregate `corpus.json`/`llms.txt` spanning every member site.
+
## [0.6.4] - 2026-07-06
- **Fixed: sites always opened in light mode until you re-picked a theme.** If you'd chosen dark (or left it on "system" with a dark device), the page still loaded light on every visit and only switched after you opened the theme menu again. The saved theme is now re-applied before the page paints, so your choice sticks across reloads — no flash, no re-toggling.