Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 0164030fd1d5816083cc035851adf00d82fe16a2
parent 403cdae39b1d27e3d8e2aed694a95221c7d0762e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 13:15:53 -0400

tags S1.4: spec 4 — /tags.json is a declared, fetchable document

corpusSpec 3 -> 4 and tags.json joins ROOT_FILES. Per the rule in corpus.ts, a
bump announces a new fetchable document, and this is one: a reader that skips
it cannot narrow a corpus to "every stream where X collabs". The per-record
`curatedTags` key is additive and bumps no manifest version.

buildSiteCorpus gains `tags: {url, videoField: "curatedTags", description}`,
emitted only when the site actually published a vocabulary, plus one llms.txt
line. The description spells out the field name, because `tags` on a record is
the platform's keywords and a machine reader has no other way to know.

compose-site writes /tags.json beside the aliases: the site's effective
vocabulary joined to the tag-counts.json build:index staged, with hidden tags
and tags with no video ON THIS SITE dropped; nothing left means no file, and
the stale copy is removed so a site that loses its last tagged video stops
advertising tags on the next build.

Four guards caught the wire change and are now updated with it: the two
service workers' hand-written ROOT_RE, the _headers CORS list and its fixture,
the offline-cache URL lists, and the contract/corpus spec assertions — which is
exactly what they are for.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Mcommon/bin/compose-site.ts | 44++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/archive/contract.test.ts | 7+++++--
Mcommon/lib/archive/contract.ts | 12+++++++++++-
Mcommon/lib/archive/headers.test.ts | 2++
Mcommon/lib/archive/headers.ts | 1+
Mcommon/lib/archive/offlineUrls.test.ts | 3+++
Mcommon/lib/corpus.test.ts | 43++++++++++++++++++++++++++++++++++---------
Mcommon/lib/corpus.ts | 37+++++++++++++++++++++++++++++++++++++
Mexport/service-worker/site-sw.js | 6+++---
Mexport/service-worker/sw-hub.js | 2+-
10 files changed, 141 insertions(+), 16 deletions(-)

diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts @@ -35,6 +35,13 @@ import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescr import { shipsPwa } from "../lib/archive/contract"; import { renderHeadersFile } from "../lib/archive/headers"; import { effectiveSiteAliases } from "../lib/aliasesStore"; +import { effectiveSiteTags } from "../lib/curatedTagsStore"; +import { TAGS_FILENAME } from "../lib/curatedTags"; +import { + TAG_COUNTS_FILENAME, + publishedTagsFrom, + type TagCountsFile, +} from "../controller/curatedTagsIndex"; import { buildSiteCorpus, renderSiteLlmsTxt, @@ -137,10 +144,16 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> { /* no digests manifest for this site */ } + // Whether this site published a curated vocabulary. Presence of the file is + // the fact — it was written (or removed) just above, by the same rule that + // decides whether any tag means anything on this site. + const hasTags = await exists(path.join(paths.exportPublicDir, TAGS_FILENAME)); + const corpus = buildSiteCorpus(descriptor, { hasArchives, postCounts, digestCounts, + hasTags, }); await writeFile( path.join(paths.exportPublicDir, "corpus.json"), @@ -751,6 +764,37 @@ async function main(): Promise<void> { JSON.stringify({ aliases }), ); + // --- curated tags (corpus vocabulary + this site's overlay + its counts) --- + // The vocabulary is read from source at compose time, like the aliases; the + // counts come from the per-site tag-counts.json build:index wrote while it + // was already streaming this site's summaries. + // + // What ships is only what means something HERE: hidden tags are dropped, and + // so is any tag with no video on this site — a corpus-wide tag that matched + // nothing here gets no chip here. Nothing left → no file at all, and the 404 + // is the empty state (the same contract duplicates.json has). The stale copy + // is removed in that case, so a site that loses its last tagged video stops + // advertising tags on the very next build. + const tagDefs = effectiveSiteTags(paths, siteId); + let tagCounts: TagCountsFile | null = null; + try { + tagCounts = JSON.parse( + await readFile( + path.join(paths.exportSitesIndexDir, siteId, TAG_COUNTS_FILENAME), + "utf8", + ), + ) as TagCountsFile; + } catch { + /* no counts staged for this site yet — treated as "nothing tagged" */ + } + const publishedTags = publishedTagsFrom(tagDefs, tagCounts); + const tagsDest = path.join(paths.exportPublicDir, TAGS_FILENAME); + if (publishedTags.tags.length > 0) { + await writeFile(tagsDest, JSON.stringify(publishedTags)); + } else { + await rm(tagsDest, { force: true }); + } + // --- duplicate-shorts report (global → site-filtered) --- // The detector writes one global duplicates.json over the whole channel pool; // each site only serves its own channels, so filter clusters to the site's diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts @@ -77,12 +77,15 @@ test("per-channel trees take a slug, flat trees do not", () => { assert.throws(() => pageUrl("posts", undefined, 1), /needs a channel slug/); }); -test("root files: the four JSON documents a reader fetches by name", () => { +test("root files: the five JSON documents a reader fetches by name", () => { assert.deepEqual([...ROOT_FILES], [ "corpus.json", "site.json", "search-aliases.json", "duplicates.json", + // Curated tags. Legitimately absent (like duplicates.json) — but unlike it, + // DECLARED in corpus.json when present, which is what corpusSpec 4 says. + "tags.json", ]); assert.equal(corpusUrl(), "/corpus.json"); assert.equal(corpusUrl("https://x.example"), "https://x.example/corpus.json"); @@ -161,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => { test("CONTRACT is frozen where it is published", () => { // These are on the wire. A change here is a change to every deployed archive's // machine contract, so it belongs in a slice that says so, not in a refactor. - assert.equal(CONTRACT.corpusSpec, 3); + assert.equal(CONTRACT.corpusSpec, 4); assert.equal(CONTRACT.siteDescriptor, 1); assert.equal(CONTRACT.manifest, 3); assert.equal(CONTRACT.transcriptsManifest, 1); diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts @@ -24,6 +24,7 @@ // dependency graph is a DAG. import { DUPLICATES_FILENAME } from "../duplicates"; +import { TAGS_FILENAME } from "../curatedTags"; // The machine-readable versions and constants of the published contract. Every // value here appears on the wire, so a change is a wire change — see the @@ -31,7 +32,7 @@ import { DUPLICATES_FILENAME } from "../duplicates"; export const CONTRACT = { // /corpus.json's own `spec`. `generator` is deliberately unversioned; see the // note in lib/corpus.ts for why a credit line did not bump the spec. - corpusSpec: 3, + corpusSpec: 4, // /site.json's `contract` (siteDescriptor.ts). siteDescriptor: 1, // The four manifest versions (manifest.ts). Each is the version field of one @@ -118,11 +119,20 @@ export function archiveUrl(base: string | undefined, p: string): string { // duplicates.json is the one that is legitimately absent: compose-site only // writes it when there is at least one publishable cluster, and corpus.json // does not declare it, so a 404 here is data, not an error. +// +// tags.json is absent the same way — the curated vocabulary, with per-site +// counts, written only when this site has at least one visible tag with a +// non-zero count. Unlike duplicates.json it IS declared in corpus.json (as +// `tags`) when present, which is what the spec-4 bump announces: a reader that +// does not know about it misses a document it could have fetched. An archive +// built before spec 4 simply 404s here, and a tag filter over it must say so +// rather than silently return nothing. export const ROOT_FILES = [ "corpus.json", "site.json", "search-aliases.json", DUPLICATES_FILENAME, + TAGS_FILENAME, ] as const; export type RootFile = (typeof ROOT_FILES)[number]; diff --git a/common/lib/archive/headers.test.ts b/common/lib/archive/headers.test.ts @@ -18,6 +18,8 @@ const SITE_HEADERS = `# Generated by compose-site.ts — do not edit by hand. Access-Control-Allow-Origin: * /duplicates.json Access-Control-Allow-Origin: * +/tags.json + Access-Control-Allow-Origin: * /summaries/* Access-Control-Allow-Origin: * /subs/* diff --git a/common/lib/archive/headers.ts b/common/lib/archive/headers.ts @@ -37,6 +37,7 @@ export const SITE_CORS_PATHS: readonly string[] = [ "/site.json", "/search-aliases.json", "/duplicates.json", + "/tags.json", "/summaries/*", "/subs/*", "/transcripts/*", diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts @@ -95,6 +95,7 @@ test("the site list is the FLAT trees plus the root files, and nothing per-chann "/site.json", "/search-aliases.json", "/duplicates.json", + "/tags.json", ]); const flat = ARCHIVE_TREES.filter(isFlatTree); @@ -146,6 +147,7 @@ test("a tree the site does not ship costs one probe and contributes nothing", as "/site.json", "/search-aliases.json", "/duplicates.json", + "/tags.json", ]); }); @@ -174,5 +176,6 @@ test("a federated origin prefixes both lists and nothing else changes", async () `${O}/site.json`, `${O}/search-aliases.json`, `${O}/duplicates.json`, + `${O}/tags.json`, ]); }); diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts @@ -159,16 +159,41 @@ test("both builders stamp the project generator, and llms.txt trails it", () => } }); -test("adding `generator` did NOT bump the corpus spec", () => { - // Guard on the reasoning, not just the number: prior bumps announced a new - // fetchable layer. An informational credit string breaks no reader, so a - // client pinned to spec 3 must keep validating. If this assertion is ever - // updated, the version-history comment in corpus.ts must justify why. - assert.equal(CORPUS_SPEC_VERSION, 3); - assert.equal( - buildSiteCorpus(descriptor(), { hasArchives: false }).spec, - 3, +test("the spec is 4, and `generator` is still not why", () => { + // Guard on the reasoning, not just the number: a bump announces a new + // FETCHABLE document. spec 4 is /tags.json. The informational credit string + // added at spec 3's time breaks no reader and did not bump anything — if + // either assertion is ever updated, the version-history comment in corpus.ts + // must justify why. + assert.equal(CORPUS_SPEC_VERSION, 4); + assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4); +}); + +test("the tags pointer is present only when the site published one", () => { + // No /tags.json → corpus.json is shaped exactly as before the bump. + const none = buildSiteCorpus(descriptor(), { hasArchives: false }); + assert.equal(none.tags, undefined); + assert.doesNotMatch(renderSiteLlmsTxt(none), /tags\.json/); + + const tagged = buildSiteCorpus( + descriptor({ siteUrl: "https://demo.example" }), + { hasArchives: false, hasTags: true }, ); + assert.equal(tagged.tags?.url, "https://demo.example/tags.json"); + // The field name is the whole point: NOT `tags`, which is yt-dlp keywords. + assert.equal(tagged.tags?.videoField, "curatedTags"); + assert.match(tagged.tags?.description ?? "", /curatedTags/); + const txt = renderSiteLlmsTxt(tagged); + assert.match(txt, /https:\/\/demo\.example\/tags\.json/); + assert.match(txt, /curatedTags/); +}); + +test("a root-relative build still names tags.json correctly", () => { + const corpus = buildSiteCorpus(descriptor(), { + hasArchives: false, + hasTags: true, + }); + assert.equal(corpus.tags?.url, "/tags.json"); }); test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => { diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -1,5 +1,6 @@ import type { PublicSiteDescriptor } from "./siteDescriptor"; import { PROJECT_GENERATOR } from "./project"; +import { TAGS_FILENAME } from "./curatedTags"; import { CONTRACT, archiveUrl, @@ -26,6 +27,12 @@ import { // (postScheme + per-channel posts manifest pointers). // v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under // the same shard scheme (digestScheme + per-channel digests manifest pointers). +// v4: curated tags — an operator-authored, cross-channel vocabulary published +// at /tags.json, with the per-record key `curatedTags` on summaries and +// transcript shards. A new FETCHABLE DOCUMENT, which is exactly what a bump +// announces: a reader that skips it cannot narrow a corpus to "every stream +// where X collabs". The per-record key itself is additive and bumps no manifest +// version — an older reader ignores an unknown field, as it always has. // // NOT a v4: the `generator` field added below is deliberately unversioned. Every // prior bump announced a new FETCHABLE LAYER — a reader that ignored it would @@ -176,6 +183,11 @@ export type SiteCorpus = { postScheme?: typeof POST_SCHEME; // Present when this site ships at least one digested video. digestScheme?: typeof DIGEST_SCHEME; + // Present when this site publishes at least one curated tag. `url` is + // /tags.json (the vocabulary + this site's per-tag counts) and `videoField` + // names the per-record key those ids appear in — deliberately NOT `tags`, + // which on a transcript record is the platform's own keywords. + tags?: { url: string; videoField: "curatedTags"; description: string }; // Present when this build ships bulk-download archives (whole-channel zips). bulkArchives?: { manifest: string; note: string }; // Pointer to the human page and BYO-key chat. @@ -211,6 +223,10 @@ export function buildSiteCorpus( // slug -> digested video count. Absent / empty means the site ships no // derived layer and digestScheme is omitted. digestCounts?: Record<string, number>; + // Whether this site published a /tags.json (compose writes one only when a + // visible tag has a non-zero count here). Absent/false leaves corpus.json + // shaped as before apart from the spec bump. + hasTags?: boolean; }, ): SiteCorpus { const base = descriptor.siteUrl; @@ -266,6 +282,19 @@ export function buildSiteCorpus( if (Object.values(digestCounts).some((n) => n > 0)) { corpus.digestScheme = DIGEST_SCHEME; } + // The curated vocabulary, when this site ships one. + if (opts.hasTags) { + corpus.tags = { + url: rootFileUrl(TAGS_FILENAME, base), + videoField: "curatedTags", + description: + "Curated cross-channel tags, applied per video by the archive's " + + "operator. Fetch tags.json for the vocabulary (id, label, group and " + + "the number of videos carrying it on this site), then filter records " + + "by their `curatedTags` array — which is NOT the same field as `tags`, " + + "the platform's own keywords. Absent on archives built before spec 4.", + }; + } if (opts.hasArchives) { corpus.bulkArchives = { manifest: join(base, "/archives/manifest.json"), @@ -353,6 +382,14 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string { `see corpus.json's digestScheme. Coverage is partial and growing.`, ); } + if (corpus.tags) { + out.push( + `- [tags.json](${corpus.tags.url}): curated cross-channel tags applied ` + + `per video by this archive's operator — the vocabulary plus how many ` + + `videos carry each one here. Records name them in \`curatedTags\` ` + + `(not \`tags\`, which is the platform's own keywords).`, + ); + } if (corpus.bulkArchives) { out.push( `- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` + diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js @@ -11,8 +11,8 @@ * PAGES — every published JSON document: the per-channel shard trees * (manifest.json + page-NNNN.json), the site-wide flat trees * (summaries, stats) and the root documents (corpus.json, site.json, - * search-aliases.json, duplicates.json). Per-channel pages are - * cache-first; per-channel manifests are network-first so a rebuild is + * search-aliases.json, duplicates.json, tags.json). Per-channel + * pages are cache-first; per-channel manifests are network-first so a rebuild is * seen, with invalidation keyed off manifest.generatedAt — when a * channel's manifest generatedAt changes, that channel's cached page * shards are evicted (the shards have stable, non-content-hashed URLs, @@ -40,7 +40,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/; // Flat trees: one manifest and its pages at the tree root, no channel level. const FLAT_RE = /^\/(summaries|stats)\/(.+)$/; // The root documents a reader fetches by name. -const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/; +const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/; self.addEventListener("install", () => { // Activate immediately — no precache list (corpus is too large to bundle). diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js @@ -33,7 +33,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/; // Flat trees: one manifest and its pages at the tree root, no channel level. const FLAT_RE = /^\/(summaries|stats)\/(.+)$/; // The root documents a reader fetches by name. -const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/; +const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/; self.addEventListener("install", () => { self.skipWaiting();