Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 04eff9b768dec55db6fbcc8c46acee6b4b2c4b96
parent 052cb600053633e48767d4b9762b1d2a1411e6f5
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 13:08:19 -0400

tags S2.7: read /tags.json — reader, cache, context, and the `tg` param

The reader side of the published document, which is a different thing from
the authoring model and now says so: lib/publishedTags.ts coerces
/tags.json as UNTRUSTED input (a hub reads one from origins it does not
control), drops anything malformed, and never throws. curatedTags.ts is
untouched.

`loadTags()` joins the ArchiveReader contract beside loadAliases, optional
for the reason duplicateIndex is, and implemented on all three transports.
A count is required per entry and a zero-count tag is not selectable: the
operator's rule is that a site offers a chip only for a tag with matches ON
THAT SITE, so there is no static list anywhere and nothing can put a chip in
front of someone that finds nothing.

`IndexedVideo` carries `curatedTags` so the filter-first page planner can
prune on it; buildVideoIndex copies it through when a summaries record has
one, which no pre-spec-4 record does.

components/tagsCache.ts mirrors aliasesCache to the letter, including the
part that matters: a status the archive ITSELF returned resolves to empty
(404 = this site publishes no tags — both "nothing curated yet" and "built
before spec 4"), while a transport failure is rethrown so react-query
retries instead of caching "no tags" for the session.

`tg=` on the URL follows `tk`'s two-letter convention and is deliberately
NOT a share-v1 key: it is the live representation of the committed
selection, so the strip-on-commit helper leaves it alone.

/tags.json is fetched via archiveUrl rather than rootFileUrl — adding it to
ROOT_FILES belongs with the corpus-spec bump that announces it, which is
S1's change, and the URL is identical either way.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Mcommon/components/SearchDataContext.tsx | 19+++++++++++++++++++
Acommon/components/tagsCache.ts | 45+++++++++++++++++++++++++++++++++++++++++++++
Mcommon/components/urlState.ts | 17+++++++++++++++++
Mcommon/lib/archive/reader-fs.ts | 17+++++++++++++++++
Mcommon/lib/archive/reader-hub.ts | 21++++++++++++++++++++-
Mcommon/lib/archive/reader.ts | 41+++++++++++++++++++++++++++++++++++++++++
Acommon/lib/publishedTags.test.ts | 123+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/publishedTags.ts | 151++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
8 files changed, 433 insertions(+), 1 deletion(-)

diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx @@ -19,6 +19,8 @@ import { useSummaries, type SummariesState } from "./summariesCache"; import { useSubsManifest } from "./subsCache"; import { usePostsManifest } from "./postsCache"; import { useSearchAliases, fetchAliases } from "./aliasesCache"; +import { useCuratedTags } from "./tagsCache"; +import type { PublishedTag } from "../lib/curatedTags"; import { mergeAliases, type SearchAlias } from "../lib/searchAliases"; import { makeId } from "./originId"; import type { DisplaySummary } from "../lib/transcripts"; @@ -68,6 +70,13 @@ export type SearchDataValue = { // Known search-alias dictionary for the leaf suggestion chip. Single-site: the // site's own list. Hub: merged across federated origins. Empty when unauthored. aliases: SearchAlias[]; + // The curated tags THIS SITE publishes, with its own per-site counts + // (/tags.json). Empty when the site publishes none — which is both "nothing + // curated yet" and "built before corpus spec 4", and in either case means the + // filter panel offers no tag chips at all. Read from the page's own origin in + // both modes: a hub's counts are the hub's, and summing member documents here + // would put a number in front of someone that no site can reproduce. + curatedTags: PublishedTag[]; }; // Single-site mode has no provenance accents; a module constant keeps the @@ -92,6 +101,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { const subsManifest = useSubsManifest("").data ?? null; const postsManifest = usePostsManifest("").data ?? null; const aliases = useSearchAliases(""); + const curatedTags = useCuratedTags(""); const manifest = summariesState.manifest; const groups = useMemo<ChannelGroup[]>(() => { @@ -139,6 +149,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { channelKeyOf: (t) => t.channel, accentOf: NO_ACCENT, aliases, + curatedTags, }), [ summariesState, @@ -148,6 +159,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) { groups, defaultGroupId, aliases, + curatedTags, ], ); @@ -382,6 +394,11 @@ export function MultiSiteDataProvider({ // eslint-disable-next-line react-hooks/exhaustive-deps }, [aliasesSettled]); + // The hub's OWN /tags.json, not a merge of its members'. See the field note + // on SearchDataValue: a federated count nobody can reproduce is worse than no + // chip, so until a hub build writes one, hub mode offers no tag chips. + const curatedTags = useCuratedTags(""); + // Synthetic merged summaries manifest. TranscriptSearch reads channels/groups // from the context (above), not from here, but the field is part of the // SummariesState contract, so provide a coherent merged view. @@ -443,6 +460,7 @@ export function MultiSiteDataProvider({ channelKeyOf: (t) => t.channelSlug, accentOf, aliases, + curatedTags, }), [ summariesState, @@ -453,6 +471,7 @@ export function MultiSiteDataProvider({ defaultGroupId, accentOf, aliases, + curatedTags, ], ); diff --git a/common/components/tagsCache.ts b/common/components/tagsCache.ts @@ -0,0 +1,45 @@ +"use client"; + +// Client fetch for a site's shipped /tags.json — the curated per-video tag +// vocabulary (lib/curatedTags.ts) with the per-site counts compose-site wrote. +// Mirrors aliasesCache / summariesCache exactly, including its answer to +// absence, because the two documents have the same shape of optionality: a +// site publishes one only when it has something to publish. +// +// A 404 is the NORMAL state here in two different ways, and neither is an +// error: compose-site writes no file at all for a site with no visible, +// non-zero-count tag, and a site built before corpus spec 4 has never heard of +// tags. Both resolve to an empty list, which the chip row reads as "offer no +// chips" — the operator's rule that a site's row is built only from what that +// site actually publishes, with no static list anywhere. + +import { useQuery } from "@tanstack/react-query"; +import { ArchiveHttpError } from "../lib/archive/reader"; +import { readerFor } from "../lib/archive/readers"; +import { coercePublishedTags } from "../lib/publishedTags"; +import type { PublishedTag } from "../lib/curatedTags"; + +const EMPTY: PublishedTag[] = []; + +// ANY status the archive itself returned resolves empty — a 404 because the +// site ships no tags, but a 500 or a 403 the same way. A TRANSPORT failure is +// different and is rethrown: it is not an answer at all, and with +// staleTime: Infinity below a swallowed blip would mean "this site has no +// tags" for the rest of the session. Same reasoning as fetchAliases. +export async function fetchTags(origin = ""): Promise<PublishedTag[]> { + try { + return coercePublishedTags(await readerFor(origin).readTags()); + } catch (err) { + if (err instanceof ArchiveHttpError) return []; + throw err; + } +} + +export function useCuratedTags(origin = ""): PublishedTag[] { + const { data } = useQuery<PublishedTag[]>({ + queryKey: ["curated-tags", origin], + queryFn: () => fetchTags(origin), + staleTime: Infinity, // static per export build + }); + return data ?? EMPTY; +} diff --git a/common/components/urlState.ts b/common/components/urlState.ts @@ -39,6 +39,17 @@ export type UrlParams = { // URL as repeated `tk=` params (e.g. ?m=subs&tk=live_chat to hide live // chat results). tracks: string[]; + // Curated tag ids selected in the filter panel, as repeated `tg=` params + // (e.g. ?tg=eva-collab&tg=eva-in-chat). ORed: a video carrying ANY of them + // passes. `tg` and not `tag`, following `tk`'s two-letter convention, and + // deliberately nowhere near the `tags` SEARCH SCOPE, which is yt-dlp + // keywords — a different thing wearing the same English word (see + // lib/curatedTags.ts). + // + // NOT a share-v1 key: unlike ch/nov/…, `tg` IS the live representation of + // the committed selection, so stripAllFilterParamsFromUrl leaves it alone + // and it round-trips through a reload on its own. + tags: string[]; }; const listeners = new Set<() => void>(); @@ -92,6 +103,7 @@ function parse(search: string): UrlParams { nu: p.get("nu") === "1", mode, tracks: p.getAll("tk"), + tags: p.getAll("tg"), }; } @@ -116,6 +128,7 @@ type Patch = Partial<{ nu: boolean; mode: SearchMode; tracks: string[]; + tags: string[]; }>; export function writeUrlParams(patch: Patch) { @@ -185,6 +198,10 @@ export function writeUrlParams(patch: Patch) { params.delete("tk"); for (const v of patch.tracks) params.append("tk", v); } + if (patch.tags !== undefined) { + params.delete("tg"); + for (const v of patch.tags) params.append("tg", v); + } const qs = params.toString(); const next = `${window.location.pathname}${qs ? `?${qs}` : ""}`; if (next === window.location.pathname + window.location.search) return; diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts @@ -23,6 +23,8 @@ import type { TranscriptDetail, DisplaySummary } from "../transcripts"; import type { SubsDetail } from "../subs"; import type { ChannelPostsManifest, Post } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import { TAGS_FILENAME, type PublishedTag } from "../curatedTags"; +import { coercePublishedTags } from "../publishedTags"; import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates"; import type { ChannelDigestsManifest, VideoDigest } from "../digests"; import type { StatsManifest, VideoStat } from "../stats"; @@ -74,6 +76,7 @@ const PREFER_PLATFORM_LINKS = process.env.TRANSCRIPT_PLATFORM_LINKS === "1"; export class LocalSource implements ArchiveReader { readonly label: string; private aliases?: SearchAlias[]; + private tags?: PublishedTag[]; private groups?: ChannelGroups; private index?: Promise<Map<string, IndexedVideo>>; private duplicates?: Promise<DuplicateIndex>; @@ -179,6 +182,20 @@ export class LocalSource implements ArchiveReader { return this.aliases; } + // A composed public dir either carries /tags.json or it does not — the same + // 404-is-data rule the remote reader follows, just as ENOENT. + async loadTags(): Promise<PublishedTag[]> { + if (this.tags) return this.tags; + try { + this.tags = coercePublishedTags( + await readLocalJson(path.join(this.dir, TAGS_FILENAME), "tags"), + ); + } catch { + this.tags = []; // no/invalid file — this build publishes no tags + } + return this.tags; + } + async loadGroups(): Promise<ChannelGroups> { if (this.groups) return this.groups; try { diff --git a/common/lib/archive/reader-hub.ts b/common/lib/archive/reader-hub.ts @@ -9,9 +9,11 @@ import type { TranscriptDetail } from "../transcripts"; import type { SubsDetail } from "../subs"; import type { ChannelPostsManifest, Post } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import { TAGS_FILENAME, type PublishedTag } from "../curatedTags"; +import { coercePublishedTags } from "../publishedTags"; import type { ChannelDigestsManifest, VideoDigest } from "../digests"; import type { VideoStat } from "../stats"; -import { corpusUrl, rootFileUrl, type HubSite } from "./contract"; +import { archiveUrl, corpusUrl, rootFileUrl, type HubSite } from "./contract"; import { EMPTY_GROUPS, RemoteSource, @@ -31,6 +33,7 @@ export class HubSource implements ArchiveReader { readonly hubBase: string; private members = new Map<string, RemoteSource>(); // siteId -> source private aliases?: SearchAlias[]; + private tags?: PublishedTag[]; private index?: Promise<Map<string, IndexedVideo>>; private duplicates?: Promise<DuplicateIndex>; private stats?: Promise<ReadonlyMap<string, VideoStat>>; @@ -104,6 +107,22 @@ export class HubSource implements ArchiveReader { return this.aliases; } + // A hub can ship its own /tags.json — the federation-wide vocabulary with + // counts summed across members. If it doesn't, tags are simply off for + // hub-wide search: merging member documents here would produce counts that + // no single site can reproduce, and a chip whose number nobody can check is + // worse than no chip. + async loadTags(): Promise<PublishedTag[]> { + if (this.tags) return this.tags; + try { + const res = await fetch(archiveUrl(this.hubBase, `/${TAGS_FILENAME}`)); + this.tags = res.ok ? coercePublishedTags(await res.json()) : []; + } catch { + this.tags = []; + } + return this.tags; + } + // In hub mode each group is itself a federated member site (a different model // — accent-per-origin), so hub-wide channel-group tokens are deferred: return // the empty fallback. Multi-channel scoping still works (member groupIds carry diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts @@ -27,6 +27,8 @@ import { summaryState, type VideoState } from "../availability"; import type { SubsDetail } from "../subs"; import type { ChannelPostsManifest, Post, PostsManifest } from "../posts"; import { coerceAliasConfig, type SearchAlias } from "../searchAliases"; +import { TAGS_FILENAME, type PublishedTag } from "../curatedTags"; +import { coercePublishedTags } from "../publishedTags"; import { resolveCanonicalSlug, DUPLICATES_FILENAME, @@ -41,6 +43,7 @@ import { type ChannelGroup, } from "../channelGroups"; import { + archiveUrl, manifestUrl, pageUrl, rootFileUrl, @@ -105,6 +108,13 @@ export type IndexedVideo = { uploadDate: string; isLivestream: boolean; ageRestricted: boolean; + // Curated tag ids (lib/curatedTags.ts), when the record carries any. Present + // here for the same reason `state` is: it is what the filter-first page + // planner prunes on, and FilterableRecord reads it. OMITTED when empty, and + // absent entirely on a site built before corpus spec 4 — which the planner + // already handles, since a video the index cannot vouch for gets its page + // read anyway. + curatedTags?: string[]; }; export type VideoIndex = ReadonlyMap<string, IndexedVideo>; @@ -273,6 +283,9 @@ export async function buildVideoIndex( uploadDate: r.uploadDate ?? "", isLivestream: r.isLivestream === true, ageRestricted: r.ageRestricted === true, + ...(Array.isArray(r.curatedTags) && r.curatedTags.length > 0 + ? { curatedTags: r.curatedTags.filter((t) => typeof t === "string") } + : {}), }); } } @@ -314,6 +327,16 @@ export interface ArchiveReader { // (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed // spellings. Result is cached per source. loadAliases(): Promise<SearchAlias[]>; + // The site's published curated tags (/tags.json) with their per-site counts. + // Returns [] when the file is absent or malformed — which is the normal + // state for a site with nothing shippable to publish AND for any site built + // before corpus spec 4, so a caller must read [] as "this source publishes no + // curated tags" and say so, never as an error. Cached per source. + // + // OPTIONAL on the interface for the same reason duplicateIndex and statsIndex + // are: the in-memory test stubs have no such concept, and a tool must degrade + // rather than assume. + loadTags?(): Promise<PublishedTag[]>; // The site's channel-group definitions (summaries/manifest.json). Returns the // empty fallback when absent/malformed, or in hub mode (federated per-site // groups are a different model — deferred). Cached per source. @@ -567,6 +590,7 @@ export class RemoteSource implements ArchiveReader { readonly label: string; private base: string; private aliases?: SearchAlias[]; + private tags?: PublishedTag[]; private groups?: ChannelGroups; private index?: Promise<Map<string, IndexedVideo>>; private duplicates?: Promise<DuplicateIndex>; @@ -701,6 +725,16 @@ export class RemoteSource implements ArchiveReader { return this.aliases; } + async loadTags(): Promise<PublishedTag[]> { + if (this.tags) return this.tags; + try { + this.tags = coercePublishedTags(await this.readTags()); + } catch { + this.tags = []; // 404 (no shippable tag, or a pre-spec-4 site) or junk + } + return this.tags; + } + async loadGroups(): Promise<ChannelGroups> { if (this.groups) return this.groups; try { @@ -762,6 +796,13 @@ export class RemoteSource implements ArchiveReader { return this.getJson<unknown>(rootFileUrl("search-aliases.json")); } + // Not a rootFileUrl() call: /tags.json joins ROOT_FILES with the corpus-spec + // bump that announces it, which is a separate change. archiveUrl is the same + // join rootFileUrl performs, so the URL is identical either way. + readTags(): Promise<unknown> { + return this.getJson<unknown>(archiveUrl(undefined, `/${TAGS_FILENAME}`)); + } + readDuplicates(): Promise<DuplicateReport> { return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME)); } diff --git a/common/lib/publishedTags.test.ts b/common/lib/publishedTags.test.ts @@ -0,0 +1,123 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + coercePublishedTags, + groupPublishedTags, + selectableTags, + tagLabels, +} from "./publishedTags"; + +const tag = (over: Record<string, unknown> = {}) => ({ + id: "eva-collab", + label: "Collab", + count: 2, + channels: { "legal-mindset": 2 }, + ...over, +}); + +test("coercePublishedTags reads the document compose-site writes", () => { + const tags = coercePublishedTags({ + version: 1, + tags: [ + tag({ group: "eva", groupLabel: "Eva", color: "#b48ead", order: 1 }), + tag({ id: "eva-topic", label: "Discussed", order: 2, count: 1 }), + ], + }); + assert.equal(tags.length, 2); + assert.deepEqual(tags.map((t) => t.id), ["eva-collab", "eva-topic"]); + assert.equal(tags[0].groupLabel, "Eva"); + assert.equal(tags[0].count, 2); + assert.deepEqual(tags[0].channels, { "legal-mindset": 2 }); +}); + +test("absence and junk both read as 'this site publishes no tags'", () => { + // The two shapes of absence a viewer actually meets: the file is a 404 (a + // site with nothing shippable, or one built before corpus spec 4) and the + // file is there but unrecognisable. Neither may throw. + for (const raw of [null, undefined, "", 42, {}, { tags: null }, { tags: {} }]) { + assert.deepEqual(coercePublishedTags(raw), []); + } +}); + +test("a malformed entry is dropped, not fatal", () => { + const tags = coercePublishedTags({ + version: 1, + tags: [ + null, + { label: "no id" }, + tag({ id: "Not A Tag Id" }), + tag({ count: undefined }), // no count — cannot be offered as a chip + tag({ count: "many" }), + tag(), + ], + }); + assert.deepEqual(tags.map((t) => t.id), ["eva-collab"]); +}); + +test("ids are lowercased and the first definition of an id wins", () => { + const tags = coercePublishedTags({ + tags: [tag({ id: "EVA-COLLAB", label: "First" }), tag({ label: "Second" })], + }); + assert.equal(tags.length, 1); + assert.equal(tags[0].id, "eva-collab"); + assert.equal(tags[0].label, "First"); +}); + +test("a tag with no label falls back to its id", () => { + const tags = coercePublishedTags({ tags: [tag({ label: " " })] }); + assert.equal(tags[0].label, "eva-collab"); +}); + +test("tags sort by order, then id — the same row on every site", () => { + const tags = coercePublishedTags({ + tags: [ + tag({ id: "c", order: 3 }), + tag({ id: "a", order: 1 }), + tag({ id: "b", order: 1 }), + ], + }); + assert.deepEqual(tags.map((t) => t.id), ["a", "b", "c"]); +}); + +test("selectableTags refuses a chip that would match nothing", () => { + const tags = coercePublishedTags({ + tags: [tag(), tag({ id: "stale", count: 0, channels: {} })], + }); + assert.deepEqual(selectableTags(tags).map((t) => t.id), ["eva-collab"]); +}); + +test("groupPublishedTags keeps tag order and sinks the ungrouped bucket", () => { + const tags = coercePublishedTags({ + tags: [ + tag({ id: "loose", order: 1 }), + tag({ id: "eva-collab", group: "eva", groupLabel: "Eva", order: 2 }), + tag({ id: "eva-topic", group: "eva", order: 3 }), + ], + }); + const groups = groupPublishedTags(tags); + assert.deepEqual( + groups.map((g) => [g.id, g.label, g.tags.map((t) => t.id)]), + [ + ["eva", "Eva", ["eva-collab", "eva-topic"]], + ["", "", ["loose"]], + ], + ); +}); + +test("a group label carried by a later member is still used", () => { + const groups = groupPublishedTags( + coercePublishedTags({ + tags: [ + tag({ id: "a", group: "eva", order: 1 }), + tag({ id: "b", group: "eva", groupLabel: "Eva", order: 2 }), + ], + }), + ); + assert.equal(groups[0].label, "Eva"); +}); + +test("tagLabels falls back to the id for a tag this site does not publish", () => { + const labels = tagLabels(coercePublishedTags({ tags: [tag()] })); + assert.equal(labels.get("eva-collab"), "Collab"); + assert.equal(labels.get("eva-topic"), undefined); +}); diff --git a/common/lib/publishedTags.ts b/common/lib/publishedTags.ts @@ -0,0 +1,151 @@ +// The READER side of the published /tags.json. +// +// `lib/curatedTags.ts` is the authoring model — defs, rules, assignments, and +// the folds the index build runs. This file is the other end of the wire: what +// a viewer, an MCP source or any other client does with the small presentation +// document compose-site writes out. It is deliberately a separate module from +// the frozen one: +// +// - the authoring model never travels to a browser (rules, regexes, +// provenance and every assignment stay server-side), and +// - a published document is UNTRUSTED input to its reader — a federated hub +// reads /tags.json from origins it does not control — so the coercion has +// to be as tolerant as coerceAliasConfig, dropping anything malformed +// rather than throwing. +// +// Absence is a legitimate empty state at every layer: compose-site writes no +// file for a site with nothing shippable (exactly like duplicates.json), and a +// site built before corpus spec 4 has never heard of tags at all. Both arrive +// here as `[]`, and every caller must treat that as "this site publishes no +// curated tags" rather than as an error. + +import type { PublishedTag, PublishedTags } from "./curatedTags"; +import { TAG_ID_RE } from "./curatedTags"; + +function str(v: unknown): string | undefined { + if (typeof v !== "string") return undefined; + const s = v.trim(); + return s === "" ? undefined : s; +} + +function num(v: unknown): number | undefined { + return typeof v === "number" && Number.isFinite(v) ? v : undefined; +} + +function coerceChannels(v: unknown): Record<string, number> { + const out: Record<string, number> = {}; + if (!v || typeof v !== "object") return out; + for (const [slug, n] of Object.entries(v as Record<string, unknown>)) { + const count = num(n); + if (!slug.trim() || count === undefined || count < 0) continue; + out[slug] = Math.floor(count); + } + return out; +} + +function coerceTag(raw: unknown): PublishedTag | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + const id = str(r.id)?.toLowerCase(); + if (!id || !TAG_ID_RE.test(id)) return null; + // A count is the one field the chip row cannot invent: the operator's rule + // is that a tag is offered only when it has matches ON THIS SITE, so a + // countless entry is dropped rather than shown as a chip that finds nothing. + const count = num(r.count); + if (count === undefined || count < 0) return null; + return { + id, + label: str(r.label) ?? id, + ...(str(r.group) ? { group: str(r.group) } : {}), + ...(str(r.groupLabel) ? { groupLabel: str(r.groupLabel) } : {}), + ...(str(r.color) ? { color: str(r.color) } : {}), + ...(num(r.order) !== undefined ? { order: num(r.order) } : {}), + count: Math.floor(count), + channels: coerceChannels(r.channels), + }; +} + +// Parse a published /tags.json into the tag list, sorted the way chips render: +// by `order` (defaulting to the document's own position) then by id, so two +// sites that ship the same vocabulary show it in the same order. NEVER throws; +// anything unrecognisable yields []. +export function coercePublishedTags(raw: unknown): PublishedTag[] { + if (!raw || typeof raw !== "object") return []; + const list = (raw as Partial<PublishedTags>).tags; + if (!Array.isArray(list)) return []; + const out: PublishedTag[] = []; + const seen = new Set<string>(); + list.forEach((entry) => { + const tag = coerceTag(entry); + if (!tag) return; + if (seen.has(tag.id)) return; // first definition of an id wins + seen.add(tag.id); + out.push(tag); + }); + const rank = new Map<string, number>(); + out.forEach((t, i) => rank.set(t.id, t.order ?? i)); + return out.sort((a, b) => { + const ra = rank.get(a.id)!; + const rb = rank.get(b.id)!; + if (ra !== rb) return ra - rb; + return a.id < b.id ? -1 : a.id > b.id ? 1 : 0; + }); +} + +// The chips the UI may offer: visible tags with at least one video ON THIS +// SITE. compose-site already drops hidden and zero-count tags, so this is a +// belt-and-braces re-check on the reader side — a hand-written or stale +// /tags.json must not be able to put a chip that matches nothing in front of +// someone. +export function selectableTags( + tags: readonly PublishedTag[], +): PublishedTag[] { + return tags.filter((t) => t.count > 0); +} + +// Group the selectable tags for a chip row, preserving the tag order within +// each group and ordering the groups by their first member. Ungrouped tags +// fall into one trailing bucket with no label, which is what a site using +// free-form ids and no `group` gets. +export type PublishedTagGroup = { + // "" for the ungrouped bucket. + id: string; + label: string; + tags: PublishedTag[]; +}; + +export function groupPublishedTags( + tags: readonly PublishedTag[], +): PublishedTagGroup[] { + const out: PublishedTagGroup[] = []; + const byId = new Map<string, PublishedTagGroup>(); + for (const tag of tags) { + const id = tag.group ?? ""; + let group = byId.get(id); + if (!group) { + group = { id, label: "", tags: [] }; + byId.set(id, group); + out.push(group); + } + // The first EXPLICIT groupLabel wins; a later member may carry the one the + // first omitted, so the id fallback is applied after the whole walk rather + // than on first sight (which would make the upgrade unreachable). + if (id && !group.label && tag.groupLabel) group.label = tag.groupLabel; + group.tags.push(tag); + } + for (const group of out) { + if (group.id && !group.label) group.label = group.id; + } + // The ungrouped bucket always sorts last: it is the leftovers, not a group. + return out.sort((a, b) => (a.id === "" ? 1 : b.id === "" ? -1 : 0)); +} + +// id -> label, for rendering a record's `curatedTags` on a card. A tag the +// site does not publish (hidden, zero-count here, or defined after this page +// was built) falls back to its id — the fact is on the record either way, and +// showing the raw id is more honest than dropping it. +export function tagLabels( + tags: readonly PublishedTag[], +): ReadonlyMap<string, string> { + return new Map(tags.map((t) => [t.id, t.label])); +}