import { test } from "node:test"; import assert from "node:assert/strict"; import { buildSiteCorpus, buildHubCorpus, renderSiteLlmsTxt, renderHubLlmsTxt, renderRobotsTxt, renderSitemapXml, CORPUS_SPEC_VERSION, } from "./corpus"; import { AI_DOC_URL, PROJECT_GENERATOR } from "./project"; import type { PublicSiteDescriptor } from "./siteDescriptor"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/corpus.test.ts function descriptor( over: Partial = {}, ): PublicSiteDescriptor { return { contract: 1, siteId: "demo", siteTitle: "Demo Archive", siteDescription: "Talks and streams.", headerTitle: "Demo", homeTagline: "", pwa: false, socialLinks: [], groups: [], defaultGroupId: "default", channels: [ { slug: "alice", name: "Alice", count: 12 }, { slug: "bob", name: "Bob", count: 3, groupId: "g1" }, ], generatedAt: "2026-07-06T00:00:00.000Z", summariesVersion: 3, ...over, }; } test("buildSiteCorpus: absolute shard URLs + totals when siteUrl is set", () => { const corpus = buildSiteCorpus( descriptor({ siteUrl: "https://demo.example/", hubUrl: "https://hub.example" }), { hasArchives: true }, ); assert.equal(corpus.spec, CORPUS_SPEC_VERSION); assert.equal(corpus.kind, "site"); assert.equal(corpus.totals.channels, 2); assert.equal(corpus.totals.videos, 15); assert.equal(corpus.site.url, "https://demo.example/"); assert.equal(corpus.site.hubUrl, "https://hub.example"); assert.equal( corpus.channels[0].manifests.transcripts, "https://demo.example/transcripts/alice/manifest.json", ); assert.equal(corpus.channels[1].groupId, "g1"); assert.ok(corpus.bulkArchives, "archives present → bulkArchives block"); assert.ok(corpus.shardScheme.description.includes("paginated")); }); test("buildSiteCorpus: root-relative URLs + no archives when unset", () => { const corpus = buildSiteCorpus(descriptor(), { hasArchives: false }); assert.equal(corpus.site.url, undefined); assert.equal( corpus.channels[0].manifests.subs, "/subs/alice/manifest.json", ); assert.equal(corpus.bulkArchives, undefined); }); test("renderSiteLlmsTxt: title, corpus link, channels", () => { const corpus = buildSiteCorpus( descriptor({ siteUrl: "https://demo.example" }), { hasArchives: true }, ); const txt = renderSiteLlmsTxt(corpus); assert.match(txt, /^# Demo Archive/); assert.match(txt, /\[corpus\.json\]\(https:\/\/demo\.example\/corpus\.json\)/); assert.match(txt, /Alice — 12 transcripts/); assert.match(txt, /Bulk archives/); }); test("buildSiteCorpus: digest layer is advertised only where it exists", () => { const corpus = buildSiteCorpus(descriptor(), { hasArchives: false, digestCounts: { alice: 7, bob: 0 }, }); // Advertised on the channel that has digests… assert.equal(corpus.channels[0].digestCount, 7); assert.equal( corpus.channels[0].manifests.digests, "/digests/alice/manifest.json", ); // …and absent, not zero-valued, on the one that doesn't. A `digests` pointer // to a manifest that was never written would send clients to a 404. assert.equal(corpus.channels[1].digestCount, undefined); assert.equal(corpus.channels[1].manifests.digests, undefined); assert.ok(corpus.digestScheme, "some digests → digestScheme block"); // The sparsity contract is the part a client must not get wrong. assert.match(corpus.digestScheme!.description, /SPARSE/); assert.match(corpus.digestScheme!.chapterTiming, /SECONDS/); }); test("buildSiteCorpus: no digests → no digestScheme, unchanged channels", () => { const corpus = buildSiteCorpus(descriptor(), { hasArchives: false }); assert.equal(corpus.digestScheme, undefined); assert.equal(corpus.channels[0].manifests.digests, undefined); assert.equal(corpus.channels[0].digestCount, undefined); }); test("renderSiteLlmsTxt: names the digest layer only when present", () => { const withDigests = renderSiteLlmsTxt( buildSiteCorpus(descriptor({ siteUrl: "https://demo.example" }), { hasArchives: false, digestCounts: { alice: 7 }, }), ); assert.match(withDigests, /AI digests: 7 of these transcripts/); const without = renderSiteLlmsTxt( buildSiteCorpus(descriptor(), { hasArchives: false }), ); assert.doesNotMatch(without, /AI digests/); }); test("buildHubCorpus: drops members with no siteUrl, links each corpus", () => { const hub = buildHubCorpus( [ { siteId: "a", siteTitle: "A", siteUrl: "https://a.example/", pwa: true }, { siteId: "b", siteTitle: "B", siteUrl: "", pwa: false }, ], { hubTitle: "The Hub", hubUrl: "https://hub.example", generatedAt: "t" }, ); assert.equal(hub.kind, "hub"); assert.equal(hub.sites.length, 1); assert.equal(hub.sites[0].corpus, "https://a.example/corpus.json"); assert.equal(hub.sites[0].siteJson, "https://a.example/site.json"); const txt = renderHubLlmsTxt(hub); assert.match(txt, /^# The Hub/); assert.match(txt, /A: https:\/\/a\.example/); }); test("both builders stamp the project generator, and llms.txt trails it", () => { const site = buildSiteCorpus(descriptor({ siteUrl: "https://demo.example" }), { hasArchives: false, }); const hub = buildHubCorpus( [{ siteId: "a", siteTitle: "A", siteUrl: "https://a.example" }], { hubTitle: "The Hub", hubUrl: "https://hub.example", generatedAt: "t" }, ); assert.equal(site.generator, PROJECT_GENERATOR); assert.equal(hub.generator, PROJECT_GENERATOR); assert.match(site.generator, /^Archilyzer \(https:\/\//); // The credit is the LAST line of both llms.txt renderings, so a human or an // LLM reading top-down finds the corpus content before the colophon. for (const txt of [renderSiteLlmsTxt(site), renderHubLlmsTxt(hub)]) { const lines = txt.trimEnd().split("\n"); assert.equal(lines[lines.length - 1], `Generated by ${PROJECT_GENERATOR}`); } }); test("Use with AI is the homepage's AI and MCP doc; the chat is the instance's /ask/", () => { // Release 16 slice DX: no site or hub has a /use-with-ai page. corpus.json // keeps the key and names the doc; llms.txt's Ask AI section is two lines. const site = buildSiteCorpus(descriptor({ siteUrl: "https://demo.example" }), { hasArchives: false, }); const hub = buildHubCorpus( [{ siteId: "a", siteTitle: "A", siteUrl: "https://a.example" }], { hubTitle: "The Hub", hubUrl: "https://hub.example", generatedAt: "t" }, ); assert.equal(site.useWithAi, AI_DOC_URL); assert.equal(hub.useWithAi, AI_DOC_URL); assert.equal(AI_DOC_URL, "https://archilyzer.pages.dev/docs/ai-and-mcp/"); const siteLines = renderSiteLlmsTxt(site).split("\n"); const siteAt = siteLines.indexOf("## Ask AI"); assert.deepEqual(siteLines.slice(siteAt + 1, siteAt + 3), [ "- [Ask AI](https://demo.example/ask/): in-browser chat (bring your own API key).", `- [Use with AI](${AI_DOC_URL}): MCP-server setup for Claude Code, Cursor, and other tools.`, ]); const hubLines = renderHubLlmsTxt(hub).split("\n"); const hubAt = hubLines.indexOf("## Ask AI"); assert.deepEqual(hubLines.slice(hubAt + 1, hubAt + 3), [ "- [Ask AI](https://hub.example/ask/): ask across the whole federation (bring your own API key).", `- [Use with AI](${AI_DOC_URL}): wire up the MCP server.`, ]); for (const txt of [renderSiteLlmsTxt(site), renderHubLlmsTxt(hub)]) { assert.doesNotMatch(txt, /use-with-ai/); } }); test("the spec is 5, and `generator` is still not why", () => { // Guard on the reasoning, not just the number: a bump announces a new // FETCHABLE document. spec 4 is /tags.json, spec 5 the reports index (and a // cited site that publishes only reports). The informational credit string // added at spec 3's time breaks no reader and did not bump anything — if // either assertion is ever updated, the version-history comment in corpus.ts // must justify why. assert.equal(CORPUS_SPEC_VERSION, 5); assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 5); }); test("the tags pointer is present only when the site published one", () => { // No /tags.json → corpus.json is shaped exactly as before the bump. const none = buildSiteCorpus(descriptor(), { hasArchives: false }); assert.equal(none.tags, undefined); assert.doesNotMatch(renderSiteLlmsTxt(none), /tags\.json/); const tagged = buildSiteCorpus( descriptor({ siteUrl: "https://demo.example" }), { hasArchives: false, hasTags: true }, ); assert.equal(tagged.tags?.url, "https://demo.example/tags.json"); // The field name is the whole point: NOT `tags`, which is yt-dlp keywords. assert.equal(tagged.tags?.videoField, "curatedTags"); assert.match(tagged.tags?.description ?? "", /curatedTags/); const txt = renderSiteLlmsTxt(tagged); assert.match(txt, /https:\/\/demo\.example\/tags\.json/); assert.match(txt, /curatedTags/); }); test("a root-relative build still names tags.json correctly", () => { const corpus = buildSiteCorpus(descriptor(), { hasArchives: false, hasTags: true, }); assert.equal(corpus.tags?.url, "/tags.json"); }); test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => { assert.match( renderRobotsTxt({ siteUrl: "https://demo.example/" }), /Sitemap: https:\/\/demo\.example\/sitemap\.xml/, ); assert.doesNotMatch(renderRobotsTxt({}), /Sitemap:/); }); test("renderSitemapXml: one loc per route", () => { const xml = renderSitemapXml({ siteUrl: "https://demo.example", routes: ["/", "/changelog"], }); assert.match(xml, /https:\/\/demo\.example\/<\/loc>/); assert.match(xml, /https:\/\/demo\.example\/changelog<\/loc>/); });