import { test } from "node:test"; import assert from "node:assert/strict"; import { ARCHIVE_TREES, PER_CHANNEL_TREES, ROOT_FILES, isFlatTree } from "./contract"; import { channelArchiveUrls, siteArchiveUrls, type ManifestReader, } from "./offlineUrls"; // THE SPLIT IS THE ASSERTION. An offline copy is two lists, and if a site-wide // document ever leaks back into the per-channel one it is not a style problem: // the per-channel list is paid per channel pinned, and the per-channel EVICT // sweep (a /// prefix match in both service workers) cannot remove // anything that is not under a channel prefix. So a flat-tree URL in the // channel list is bytes that are downloaded N times and never removable. // // Measured on jeralyzer when that was the shipped behaviour: summaries ~14 MB + // stats ~21.6 MB + duplicates 5.9 MB = ~41.5 MB per channel, re-fetched for // real (the worker bulk-caches with `cache: "reload"`). // A stub archive: manifests present at these URLs with these page counts, // everything else absent. Records the reads so "absent costs one probe" is // assertable. function reader(pageCounts: Record): { read: ManifestReader; reads: string[]; } { const reads: string[] = []; const read: ManifestReader = async (url) => { reads.push(url); return url in pageCounts ? { pageCount: pageCounts[url] } : null; }; return { read, reads }; } const FULL = { "/transcripts/alpha/manifest.json": 2, "/subs/alpha/manifest.json": 1, "/posts/alpha/manifest.json": 1, "/digests/alpha/manifest.json": 1, "/summaries/manifest.json": 2, "/stats/manifest.json": 1, }; test("the channel list is the PER-CHANNEL trees and nothing else", async () => { const { read } = reader(FULL); const urls = await channelArchiveUrls(read, "", "alpha"); assert.deepEqual(urls, [ "/transcripts/alpha/manifest.json", "/transcripts/alpha/page-0000.json", "/transcripts/alpha/page-0001.json", "/subs/alpha/manifest.json", "/subs/alpha/page-0000.json", "/posts/alpha/manifest.json", "/posts/alpha/page-0000.json", "/digests/alpha/manifest.json", "/digests/alpha/page-0000.json", ]); // Stated structurally as well as literally, so adding a layer to the contract // fails here rather than silently shipping a channel that is half offline. const trees = new Set(urls.map((u) => u.split("/")[1])); assert.deepEqual([...trees].sort(), [...PER_CHANNEL_TREES].sort()); // Not one site-wide byte. This is the regression the review caught. for (const u of urls) { const tree = u.split("/")[1]; assert.ok( !ARCHIVE_TREES.some((t) => t === tree && isFlatTree(t)), `flat tree ${tree} must not be in a per-channel download: ${u}`, ); } for (const file of ROOT_FILES) { assert.ok(!urls.includes(`/${file}`), `${file} must not be per channel`); } // Every URL is under a /// prefix — the only shape the service // workers' per-channel evict sweep can ever remove. for (const u of urls) { assert.match(u, /^\/[^/]+\/alpha\//, `${u} is not evictable per channel`); } }); test("the site list is the FLAT trees plus the root files, and nothing per-channel", async () => { const { read } = reader(FULL); const urls = await siteArchiveUrls(read, ""); assert.deepEqual(urls, [ "/summaries/manifest.json", "/summaries/page-0000.json", "/summaries/page-0001.json", "/stats/manifest.json", "/stats/page-0000.json", "/corpus.json", "/site.json", "/search-aliases.json", "/duplicates.json", "/tags.json", ]); const flat = ARCHIVE_TREES.filter(isFlatTree); for (const tree of flat) { assert.ok( urls.some((u) => u.startsWith(`/${tree}/`)), `${tree} missing from the site list`, ); } for (const file of ROOT_FILES) assert.ok(urls.includes(`/${file}`)); // No channel slug anywhere: nothing here is paid per channel. for (const u of urls) assert.doesNotMatch(u, /alpha/); }); test("together the two lists cover every tree the contract defines, once", async () => { const { read } = reader(FULL); const all = [ ...(await channelArchiveUrls(read, "", "alpha")), ...(await siteArchiveUrls(read, "")), ]; assert.equal(new Set(all).size, all.length, "no URL is in both lists"); const trees = new Set( all.filter((u) => u.split("/").length > 2).map((u) => u.split("/")[1]), ); assert.deepEqual([...trees].sort(), [...ARCHIVE_TREES].sort()); }); test("a tree the site does not ship costs one probe and contributes nothing", async () => { // Transcripts only — the shape of a site with no chat, no posts, no digests // and no stats. const { read, reads } = reader({ "/transcripts/alpha/manifest.json": 1 }); assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), [ "/transcripts/alpha/manifest.json", "/transcripts/alpha/page-0000.json", ]); // One probe per absent tree, not a retry storm. assert.deepEqual(reads, [ "/transcripts/alpha/manifest.json", "/subs/alpha/manifest.json", "/posts/alpha/manifest.json", "/digests/alpha/manifest.json", ]); // The root files are listed unprobed — absent ones are skipped by the worker. assert.deepEqual(await siteArchiveUrls(read, ""), [ "/corpus.json", "/site.json", "/search-aliases.json", "/duplicates.json", "/tags.json", ]); }); test("no transcripts manifest means no download at all", async () => { const { read } = reader({ "/subs/alpha/manifest.json": 1 }); // A channel whose transcripts are unreachable cannot be opened offline, so // caching its live chat would be caching a dead end. assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), []); }); test("a federated origin prefixes both lists and nothing else changes", async () => { const O = "https://member.example"; const { read } = reader({ [`${O}/transcripts/alpha/manifest.json`]: 1, [`${O}/summaries/manifest.json`]: 1, }); assert.deepEqual(await channelArchiveUrls(read, O, "alpha"), [ `${O}/transcripts/alpha/manifest.json`, `${O}/transcripts/alpha/page-0000.json`, ]); assert.deepEqual(await siteArchiveUrls(read, O), [ `${O}/summaries/manifest.json`, `${O}/summaries/page-0000.json`, `${O}/corpus.json`, `${O}/site.json`, `${O}/search-aliases.json`, `${O}/duplicates.json`, `${O}/tags.json`, ]); });