commit 0164030fd1d5816083cc035851adf00d82fe16a2
parent 403cdae39b1d27e3d8e2aed694a95221c7d0762e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 13:15:53 -0400
tags S1.4: spec 4 — /tags.json is a declared, fetchable document
corpusSpec 3 -> 4 and tags.json joins ROOT_FILES. Per the rule in corpus.ts, a
bump announces a new fetchable document, and this is one: a reader that skips
it cannot narrow a corpus to "every stream where X collabs". The per-record
`curatedTags` key is additive and bumps no manifest version.
buildSiteCorpus gains `tags: {url, videoField: "curatedTags", description}`,
emitted only when the site actually published a vocabulary, plus one llms.txt
line. The description spells out the field name, because `tags` on a record is
the platform's keywords and a machine reader has no other way to know.
compose-site writes /tags.json beside the aliases: the site's effective
vocabulary joined to the tag-counts.json build:index staged, with hidden tags
and tags with no video ON THIS SITE dropped; nothing left means no file, and
the stale copy is removed so a site that loses its last tagged video stops
advertising tags on the next build.
Four guards caught the wire change and are now updated with it: the two
service workers' hand-written ROOT_RE, the _headers CORS list and its fixture,
the offline-cache URL lists, and the contract/corpus spec assertions — which is
exactly what they are for.
Co-Authored-By: Claude Opus <noreply@anthropic.com>
Diffstat:
10 files changed, 141 insertions(+), 16 deletions(-)
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -35,6 +35,13 @@ import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescr
import { shipsPwa } from "../lib/archive/contract";
import { renderHeadersFile } from "../lib/archive/headers";
import { effectiveSiteAliases } from "../lib/aliasesStore";
+import { effectiveSiteTags } from "../lib/curatedTagsStore";
+import { TAGS_FILENAME } from "../lib/curatedTags";
+import {
+ TAG_COUNTS_FILENAME,
+ publishedTagsFrom,
+ type TagCountsFile,
+} from "../controller/curatedTagsIndex";
import {
buildSiteCorpus,
renderSiteLlmsTxt,
@@ -137,10 +144,16 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
/* no digests manifest for this site */
}
+ // Whether this site published a curated vocabulary. Presence of the file is
+ // the fact — it was written (or removed) just above, by the same rule that
+ // decides whether any tag means anything on this site.
+ const hasTags = await exists(path.join(paths.exportPublicDir, TAGS_FILENAME));
+
const corpus = buildSiteCorpus(descriptor, {
hasArchives,
postCounts,
digestCounts,
+ hasTags,
});
await writeFile(
path.join(paths.exportPublicDir, "corpus.json"),
@@ -751,6 +764,37 @@ async function main(): Promise<void> {
JSON.stringify({ aliases }),
);
+ // --- curated tags (corpus vocabulary + this site's overlay + its counts) ---
+ // The vocabulary is read from source at compose time, like the aliases; the
+ // counts come from the per-site tag-counts.json build:index wrote while it
+ // was already streaming this site's summaries.
+ //
+ // What ships is only what means something HERE: hidden tags are dropped, and
+ // so is any tag with no video on this site — a corpus-wide tag that matched
+ // nothing here gets no chip here. Nothing left → no file at all, and the 404
+ // is the empty state (the same contract duplicates.json has). The stale copy
+ // is removed in that case, so a site that loses its last tagged video stops
+ // advertising tags on the very next build.
+ const tagDefs = effectiveSiteTags(paths, siteId);
+ let tagCounts: TagCountsFile | null = null;
+ try {
+ tagCounts = JSON.parse(
+ await readFile(
+ path.join(paths.exportSitesIndexDir, siteId, TAG_COUNTS_FILENAME),
+ "utf8",
+ ),
+ ) as TagCountsFile;
+ } catch {
+ /* no counts staged for this site yet — treated as "nothing tagged" */
+ }
+ const publishedTags = publishedTagsFrom(tagDefs, tagCounts);
+ const tagsDest = path.join(paths.exportPublicDir, TAGS_FILENAME);
+ if (publishedTags.tags.length > 0) {
+ await writeFile(tagsDest, JSON.stringify(publishedTags));
+ } else {
+ await rm(tagsDest, { force: true });
+ }
+
// --- duplicate-shorts report (global → site-filtered) ---
// The detector writes one global duplicates.json over the whole channel pool;
// each site only serves its own channels, so filter clusters to the site's
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -77,12 +77,15 @@ test("per-channel trees take a slug, flat trees do not", () => {
assert.throws(() => pageUrl("posts", undefined, 1), /needs a channel slug/);
});
-test("root files: the four JSON documents a reader fetches by name", () => {
+test("root files: the five JSON documents a reader fetches by name", () => {
assert.deepEqual([...ROOT_FILES], [
"corpus.json",
"site.json",
"search-aliases.json",
"duplicates.json",
+ // Curated tags. Legitimately absent (like duplicates.json) — but unlike it,
+ // DECLARED in corpus.json when present, which is what corpusSpec 4 says.
+ "tags.json",
]);
assert.equal(corpusUrl(), "/corpus.json");
assert.equal(corpusUrl("https://x.example"), "https://x.example/corpus.json");
@@ -161,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => {
test("CONTRACT is frozen where it is published", () => {
// These are on the wire. A change here is a change to every deployed archive's
// machine contract, so it belongs in a slice that says so, not in a refactor.
- assert.equal(CONTRACT.corpusSpec, 3);
+ assert.equal(CONTRACT.corpusSpec, 4);
assert.equal(CONTRACT.siteDescriptor, 1);
assert.equal(CONTRACT.manifest, 3);
assert.equal(CONTRACT.transcriptsManifest, 1);
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -24,6 +24,7 @@
// dependency graph is a DAG.
import { DUPLICATES_FILENAME } from "../duplicates";
+import { TAGS_FILENAME } from "../curatedTags";
// The machine-readable versions and constants of the published contract. Every
// value here appears on the wire, so a change is a wire change — see the
@@ -31,7 +32,7 @@ import { DUPLICATES_FILENAME } from "../duplicates";
export const CONTRACT = {
// /corpus.json's own `spec`. `generator` is deliberately unversioned; see the
// note in lib/corpus.ts for why a credit line did not bump the spec.
- corpusSpec: 3,
+ corpusSpec: 4,
// /site.json's `contract` (siteDescriptor.ts).
siteDescriptor: 1,
// The four manifest versions (manifest.ts). Each is the version field of one
@@ -118,11 +119,20 @@ export function archiveUrl(base: string | undefined, p: string): string {
// duplicates.json is the one that is legitimately absent: compose-site only
// writes it when there is at least one publishable cluster, and corpus.json
// does not declare it, so a 404 here is data, not an error.
+//
+// tags.json is absent the same way — the curated vocabulary, with per-site
+// counts, written only when this site has at least one visible tag with a
+// non-zero count. Unlike duplicates.json it IS declared in corpus.json (as
+// `tags`) when present, which is what the spec-4 bump announces: a reader that
+// does not know about it misses a document it could have fetched. An archive
+// built before spec 4 simply 404s here, and a tag filter over it must say so
+// rather than silently return nothing.
export const ROOT_FILES = [
"corpus.json",
"site.json",
"search-aliases.json",
DUPLICATES_FILENAME,
+ TAGS_FILENAME,
] as const;
export type RootFile = (typeof ROOT_FILES)[number];
diff --git a/common/lib/archive/headers.test.ts b/common/lib/archive/headers.test.ts
@@ -18,6 +18,8 @@ const SITE_HEADERS = `# Generated by compose-site.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/duplicates.json
Access-Control-Allow-Origin: *
+/tags.json
+ Access-Control-Allow-Origin: *
/summaries/*
Access-Control-Allow-Origin: *
/subs/*
diff --git a/common/lib/archive/headers.ts b/common/lib/archive/headers.ts
@@ -37,6 +37,7 @@ export const SITE_CORS_PATHS: readonly string[] = [
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
"/summaries/*",
"/subs/*",
"/transcripts/*",
diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts
@@ -95,6 +95,7 @@ test("the site list is the FLAT trees plus the root files, and nothing per-chann
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
const flat = ARCHIVE_TREES.filter(isFlatTree);
@@ -146,6 +147,7 @@ test("a tree the site does not ship costs one probe and contributes nothing", as
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
});
@@ -174,5 +176,6 @@ test("a federated origin prefixes both lists and nothing else changes", async ()
`${O}/site.json`,
`${O}/search-aliases.json`,
`${O}/duplicates.json`,
+ `${O}/tags.json`,
]);
});
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -159,16 +159,41 @@ test("both builders stamp the project generator, and llms.txt trails it", () =>
}
});
-test("adding `generator` did NOT bump the corpus spec", () => {
- // Guard on the reasoning, not just the number: prior bumps announced a new
- // fetchable layer. An informational credit string breaks no reader, so a
- // client pinned to spec 3 must keep validating. If this assertion is ever
- // updated, the version-history comment in corpus.ts must justify why.
- assert.equal(CORPUS_SPEC_VERSION, 3);
- assert.equal(
- buildSiteCorpus(descriptor(), { hasArchives: false }).spec,
- 3,
+test("the spec is 4, and `generator` is still not why", () => {
+ // Guard on the reasoning, not just the number: a bump announces a new
+ // FETCHABLE document. spec 4 is /tags.json. The informational credit string
+ // added at spec 3's time breaks no reader and did not bump anything — if
+ // either assertion is ever updated, the version-history comment in corpus.ts
+ // must justify why.
+ assert.equal(CORPUS_SPEC_VERSION, 4);
+ assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4);
+});
+
+test("the tags pointer is present only when the site published one", () => {
+ // No /tags.json → corpus.json is shaped exactly as before the bump.
+ const none = buildSiteCorpus(descriptor(), { hasArchives: false });
+ assert.equal(none.tags, undefined);
+ assert.doesNotMatch(renderSiteLlmsTxt(none), /tags\.json/);
+
+ const tagged = buildSiteCorpus(
+ descriptor({ siteUrl: "https://demo.example" }),
+ { hasArchives: false, hasTags: true },
);
+ assert.equal(tagged.tags?.url, "https://demo.example/tags.json");
+ // The field name is the whole point: NOT `tags`, which is yt-dlp keywords.
+ assert.equal(tagged.tags?.videoField, "curatedTags");
+ assert.match(tagged.tags?.description ?? "", /curatedTags/);
+ const txt = renderSiteLlmsTxt(tagged);
+ assert.match(txt, /https:\/\/demo\.example\/tags\.json/);
+ assert.match(txt, /curatedTags/);
+});
+
+test("a root-relative build still names tags.json correctly", () => {
+ const corpus = buildSiteCorpus(descriptor(), {
+ hasArchives: false,
+ hasTags: true,
+ });
+ assert.equal(corpus.tags?.url, "/tags.json");
});
test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => {
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -1,5 +1,6 @@
import type { PublicSiteDescriptor } from "./siteDescriptor";
import { PROJECT_GENERATOR } from "./project";
+import { TAGS_FILENAME } from "./curatedTags";
import {
CONTRACT,
archiveUrl,
@@ -26,6 +27,12 @@ import {
// (postScheme + per-channel posts manifest pointers).
// v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under
// the same shard scheme (digestScheme + per-channel digests manifest pointers).
+// v4: curated tags — an operator-authored, cross-channel vocabulary published
+// at /tags.json, with the per-record key `curatedTags` on summaries and
+// transcript shards. A new FETCHABLE DOCUMENT, which is exactly what a bump
+// announces: a reader that skips it cannot narrow a corpus to "every stream
+// where X collabs". The per-record key itself is additive and bumps no manifest
+// version — an older reader ignores an unknown field, as it always has.
//
// NOT a v4: the `generator` field added below is deliberately unversioned. Every
// prior bump announced a new FETCHABLE LAYER — a reader that ignored it would
@@ -176,6 +183,11 @@ export type SiteCorpus = {
postScheme?: typeof POST_SCHEME;
// Present when this site ships at least one digested video.
digestScheme?: typeof DIGEST_SCHEME;
+ // Present when this site publishes at least one curated tag. `url` is
+ // /tags.json (the vocabulary + this site's per-tag counts) and `videoField`
+ // names the per-record key those ids appear in — deliberately NOT `tags`,
+ // which on a transcript record is the platform's own keywords.
+ tags?: { url: string; videoField: "curatedTags"; description: string };
// Present when this build ships bulk-download archives (whole-channel zips).
bulkArchives?: { manifest: string; note: string };
// Pointer to the human page and BYO-key chat.
@@ -211,6 +223,10 @@ export function buildSiteCorpus(
// slug -> digested video count. Absent / empty means the site ships no
// derived layer and digestScheme is omitted.
digestCounts?: Record<string, number>;
+ // Whether this site published a /tags.json (compose writes one only when a
+ // visible tag has a non-zero count here). Absent/false leaves corpus.json
+ // shaped as before apart from the spec bump.
+ hasTags?: boolean;
},
): SiteCorpus {
const base = descriptor.siteUrl;
@@ -266,6 +282,19 @@ export function buildSiteCorpus(
if (Object.values(digestCounts).some((n) => n > 0)) {
corpus.digestScheme = DIGEST_SCHEME;
}
+ // The curated vocabulary, when this site ships one.
+ if (opts.hasTags) {
+ corpus.tags = {
+ url: rootFileUrl(TAGS_FILENAME, base),
+ videoField: "curatedTags",
+ description:
+ "Curated cross-channel tags, applied per video by the archive's " +
+ "operator. Fetch tags.json for the vocabulary (id, label, group and " +
+ "the number of videos carrying it on this site), then filter records " +
+ "by their `curatedTags` array — which is NOT the same field as `tags`, " +
+ "the platform's own keywords. Absent on archives built before spec 4.",
+ };
+ }
if (opts.hasArchives) {
corpus.bulkArchives = {
manifest: join(base, "/archives/manifest.json"),
@@ -353,6 +382,14 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
`see corpus.json's digestScheme. Coverage is partial and growing.`,
);
}
+ if (corpus.tags) {
+ out.push(
+ `- [tags.json](${corpus.tags.url}): curated cross-channel tags applied ` +
+ `per video by this archive's operator — the vocabulary plus how many ` +
+ `videos carry each one here. Records name them in \`curatedTags\` ` +
+ `(not \`tags\`, which is the platform's own keywords).`,
+ );
+ }
if (corpus.bulkArchives) {
out.push(
`- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` +
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -11,8 +11,8 @@
* PAGES — every published JSON document: the per-channel shard trees
* (manifest.json + page-NNNN.json), the site-wide flat trees
* (summaries, stats) and the root documents (corpus.json, site.json,
- * search-aliases.json, duplicates.json). Per-channel pages are
- * cache-first; per-channel manifests are network-first so a rebuild is
+ * search-aliases.json, duplicates.json, tags.json). Per-channel
+ * pages are cache-first; per-channel manifests are network-first so a rebuild is
* seen, with invalidation keyed off manifest.generatedAt — when a
* channel's manifest generatedAt changes, that channel's cached page
* shards are evicted (the shards have stable, non-content-hashed URLs,
@@ -40,7 +40,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js
@@ -33,7 +33,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
self.skipWaiting();