// WHAT AN OFFLINE COPY OF AN ARCHIVE CONSISTS OF, split the way it is paid for. // // Two lists, and the split is the whole point of this module: // // channelArchiveUrls — the PER-CHANNEL trees of one channel. Pinning a second // channel genuinely costs this again, because none of it // is shared. // siteArchiveUrls — the SITE-WIDE documents of one origin: the flat trees // (summaries, stats) and the root files. Identical for // every channel of that origin, so it is downloaded once // and evicted when the last pinned channel goes. // // The first cut of this shipped as ONE list and it was a real cost, not a // tidiness point. Measured on jeralyzer: summaries ~14 MB, stats ~21.6 MB (its // page-0000 alone is 20.97 MB) and /duplicates.json 5.9 MB — ~41.5 MB of // byte-identical site data re-fetched per channel pinned (the service worker // bulk-fetches with `cache: "reload"`, so they really do go over the wire), and // the per-channel evict path sweeps only the per-channel prefixes, so removing // a channel left every byte of it behind forever. // // The manifest READER IS INJECTED. The caller owns the transport, which matters // here: the viewer reads these with `cache: "no-store"` precisely because a // normal fetch would be answered from the service-worker cache this list exists // to refill, and would then enumerate the stale copy. Injection also makes the // lists directly assertable without a network. import { ARCHIVE_TREES, PER_CHANNEL_TREES, ROOT_FILES, isFlatTree, manifestUrl, pageUrl, rootFileUrl, } from "./contract"; // Every manifest shape these walks need, reduced to the one field a URL list is // built from. A tree a site does not ship answers 404 and reads as absent. export type PagedManifest = { pageCount?: number }; // Reads one manifest by URL. Resolves null for "this site does not ship it", // which for these lists is normal rather than an error. export type ManifestReader = (url: string) => Promise; // The manifest plus every page of one tree, or [] when the tree is absent. // `slug` is undefined for a flat tree. async function treeUrls( read: ManifestReader, base: string, tree: (typeof ARCHIVE_TREES)[number], slug: string | undefined, ): Promise { const manifest = manifestUrl(tree, slug, base); const m = await read(manifest); if (!m) return []; const urls = [manifest]; for (let p = 0; p < (m.pageCount ?? 0); p++) { urls.push(pageUrl(tree, slug, p, base)); } return urls; } // One channel's shards, across every per-channel tree the contract defines. // // Returns [] when the channel has no TRANSCRIPTS manifest — that is not a tree // a readable channel can be missing, so the caller refuses the download rather // than caching a channel that cannot be opened. Every other tree is optional // and costs one 404 when absent. export async function channelArchiveUrls( read: ManifestReader, base: string, slug: string, ): Promise { const transcripts = await treeUrls(read, base, "transcripts", slug); if (transcripts.length === 0) return []; const urls = [...transcripts]; for (const tree of PER_CHANNEL_TREES) { if (tree === "transcripts") continue; urls.push(...(await treeUrls(read, base, tree, slug))); } return urls; } // One origin's site-wide documents: the flat trees and the root files. // // The root files are listed unconditionally rather than probed. /duplicates.json // and /search-aliases.json are legitimately absent on many sites, the caller // hands the list to a worker that skips what it cannot fetch, and probing them // first would double the request count to learn something the fetch already // tells us. export async function siteArchiveUrls( read: ManifestReader, base: string, ): Promise { const urls: string[] = []; for (const tree of ARCHIVE_TREES) { if (!isFlatTree(tree)) continue; urls.push(...(await treeUrls(read, base, tree, undefined))); } for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base)); return urls; }