commit 5eb761be296b90659bf2693096d7b27b1a7d73d3
parent c70b355a074669b0c18de909a8cc925434b67e4b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 12 Sep 2026 02:45:43 -0400
export: offline downloads every layer the contract has, and the SWs serve them
The offline download list was transcripts-only. That is why the duplicates
page was dead with no connection — /duplicates.json is a root document that
nothing downloaded — and why an offline channel had no live chat, no social
posts and no AI digests. It now enumerates ARCHIVE_TREES plus ROOT_FILES, so
the next layer added to the contract is offline-capable the day it ships
rather than the day someone notices.
Downloading is only half of it, and the half that was missing is the reason
this is one commit rather than two: the site SW's SHARD_RE matches three path
segments, so /summaries/manifest.json and /duplicates.json never matched it
and the fetch handler let them straight through to the network. A downloaded
root file sat in the cache and was never served from it. Both workers gain
FLAT_RE and ROOT_RE alongside SHARD_RE, answered network-first — those
documents have no per-channel generatedAt to evict against, so cache-first
would be stale until the SW version moved.
Three drifts fixed while the lists were being written down. `summaries` was
in site-sw's SHARD_RE, which is dead: it is a flat tree, so /summaries/<slug>/
does not exist. The same dead prefix sat in both eviction lists, while
/posts/ and /digests/ were absent from them — a rebuilt channel kept serving
its old posts and digests from cache. And sw-hub had no `digests` at all: the
hub federated a layer it could never cache.
A service worker cannot import, so those lists stay hand-written — and
contract.test.ts now reads both files off disk and fails when their tree and
root-file sets drift from CONTRACT. Same for the search index worker's
deliberate private copy of pageFileName, which is pinned to CONTRACT.pagePad
by extracting and checking the padding it actually uses. Verified by
mutation: dropping `digests` from sw-hub's SHARD_RE turns the suite red.
offlineCache keeps its own `cache: "no-store"` fetch rather than taking the
shared reader. The reader is a normal fetch, so it would be answered from the
very service-worker cache this function exists to refill, and would compute
its download list from the stale copy. What is shared is the thing that
actually drifted: the URL shape.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
4 files changed, 249 insertions(+), 33 deletions(-)
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -1,7 +1,12 @@
import { test } from "node:test";
import assert from "node:assert/strict";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
import {
+ ARCHIVE_TREES,
CONTRACT,
+ PER_CHANNEL_TREES,
ROOT_FILES,
archiveUrl,
corpusUrl,
@@ -15,6 +20,8 @@ import {
import { buildSiteCorpus } from "../corpus";
import type { PublicSiteDescriptor } from "../siteDescriptor";
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+
// A minimal descriptor with two channels, one of which ships posts and digests,
// so every conditional manifest pointer buildSiteCorpus can emit is exercised.
function descriptor(siteUrl?: string): PublicSiteDescriptor {
@@ -166,3 +173,88 @@ test("CONTRACT is frozen where it is published", () => {
["transcripts", "subs", "posts", "digests", "summaries"],
);
});
+
+// ─── The three hand-written copies of the contract, pinned ───
+//
+// Three files legitimately cannot import this module and re-spell part of it by
+// hand: the two service workers (a service worker has no module graph to import
+// through — it is fetched and evaluated standalone) and the search index worker
+// (a classic Worker bundle). Each copy is silent when it drifts: a layer the SW
+// does not match is simply never cached, and a page file name without the
+// padding 404s against a real site while passing every unit test that made up
+// its own fixture names.
+//
+// So the copies are read off disk and compared with the contract. This is the
+// guard that lets CONTRACT.layers grow: add a layer, and these fail by name.
+
+const REPO = path.resolve(HERE, "..", "..", "..");
+
+function readSource(rel: string): string {
+ return readFileSync(path.join(REPO, rel), "utf8");
+}
+
+// Pull `const <NAME> = /…/;` out of a service worker and return the alternatives
+// of its first (…) group, un-escaped. Throws rather than returning [] when the
+// constant is missing, so a rename cannot make this test vacuously pass.
+function alternatives(src: string, name: string): string[] {
+ const decl = new RegExp(`const ${name} = /(.+)/;`).exec(src);
+ assert.ok(decl, `${name} not found — did it move or get renamed?`);
+ const group = /\(([^)]+)\)/.exec(decl[1]);
+ assert.ok(group, `${name} has no alternation group`);
+ return group[1].split("|").map((a) => a.replace(/\\(.)/g, "$1"));
+}
+
+// The `/<tree>/${slug}/` entries of an eviction prefix list.
+function evictedTrees(src: string): string[] {
+ const block = /const prefixes = \[([\s\S]*?)\]/.exec(src);
+ assert.ok(block, "eviction prefix list not found");
+ return [...block[1].matchAll(/`\/([^/]+)\/\$\{slug\}\/`/g)].map((m) => m[1]);
+}
+
+for (const sw of ["export/service-worker/site-sw.js", "export/service-worker/sw-hub.js"]) {
+ test(`${sw} matches exactly the contract's trees and root files`, () => {
+ const src = readSource(sw);
+
+ // Per-channel and flat, split the way the URL shapes are split — and
+ // together exactly ARCHIVE_TREES, so a layer cannot be quietly dropped from
+ // one family and "found" in the other.
+ assert.deepEqual(
+ alternatives(src, "SHARD_RE").sort(),
+ [...PER_CHANNEL_TREES].sort(),
+ );
+ assert.deepEqual(
+ alternatives(src, "FLAT_RE").sort(),
+ ARCHIVE_TREES.filter(isFlatTree).sort(),
+ );
+ assert.deepEqual(
+ [...alternatives(src, "SHARD_RE"), ...alternatives(src, "FLAT_RE")].sort(),
+ [...ARCHIVE_TREES].sort(),
+ );
+
+ // The root documents, by name. ROOT_RE is a full-path match, so these are
+ // the file names with no extra path.
+ assert.deepEqual(alternatives(src, "ROOT_RE").sort(), [...ROOT_FILES].sort());
+
+ // Eviction is per channel, so it sweeps the per-channel trees and nothing
+ // else. A flat tree here would be a prefix that can never match.
+ assert.deepEqual(evictedTrees(src).sort(), [...PER_CHANNEL_TREES].sort());
+ });
+}
+
+test("the search index worker's private pageFileName matches CONTRACT.pagePad", () => {
+ // components/searchIndex.worker.ts keeps its own copy on purpose — a worker
+ // bundle cannot import this module. Extract the literal padding it uses and
+ // run the copy, so both the constant AND the produced name are pinned.
+ const src = readSource("common/components/searchIndex.worker.ts");
+ const fn = /function pageFileName\(index: number\): string \{\s*return `page-\$\{String\(index\)\.padStart\((\d+), "0"\)\}\.json`;\s*\}/.exec(
+ src,
+ );
+ assert.ok(fn, "searchIndex.worker.ts's pageFileName is not the shape this test pins");
+ assert.equal(Number(fn[1]), CONTRACT.pagePad);
+
+ // And the URLs it builds are the contract's, for the one tree it walks.
+ assert.ok(src.includes("`/transcripts/${slug}/manifest.json`"));
+ assert.equal(manifestUrl("transcripts", "alpha"), "/transcripts/alpha/manifest.json");
+ assert.ok(src.includes("`/transcripts/${slug}/${pageFileName(p)}`"));
+ assert.equal(pageUrl("transcripts", "alpha", 12), "/transcripts/alpha/page-0012.json");
+});
diff --git a/export/app/lib/offlineCache.ts b/export/app/lib/offlineCache.ts
@@ -12,9 +12,14 @@
// per (origin, slug); the site SW ignores it (same-origin only).
import {
- pageFileName,
- type ChannelTranscriptsManifest,
-} from "yt-dlp-transcript-common/lib/manifest";
+ ARCHIVE_TREES,
+ ROOT_FILES,
+ isFlatTree,
+ manifestUrl,
+ pageUrl,
+ rootFileUrl,
+ type ArchiveTree,
+} from "yt-dlp-transcript-common/lib/archive/contract";
import { idBaseUrl, makeId } from "yt-dlp-transcript-common/components/originId";
const PINNED_KEY = "ytdlp-tb:offline-channels";
@@ -56,26 +61,60 @@ function sendToSw(
});
}
-async function fetchManifest(
- origin: string,
- slug: string,
-): Promise<ChannelTranscriptsManifest | null> {
+// Every manifest shape the walk needs, reduced to the one field a URL list is
+// built from. A tree a site does not ship answers 404 and reads as absent.
+type PagedManifest = { pageCount?: number };
+
+// Deliberately NOT the shared ArchiveReader: the reader is a normal `fetch`, so
+// a page walking it would be answered from the very service-worker cache this
+// function exists to REFILL, and would compute its download list from the stale
+// copy. `cache: "no-store"` is the whole point of this read. What is shared is
+// the thing that actually drifted — the URL shape.
+async function readManifest(url: string): Promise<PagedManifest | null> {
try {
- const res = await fetch(`${idBaseUrl(origin)}/transcripts/${slug}/manifest.json`, {
- cache: "no-store",
- });
+ const res = await fetch(url, { cache: "no-store" });
if (!res.ok) return null;
- return (await res.json()) as ChannelTranscriptsManifest;
+ return (await res.json()) as PagedManifest;
} catch {
return null;
}
}
+// The manifest + every page of one tree, or [] when the site ships no such
+// tree. `slug` is undefined for the flat trees (summaries, stats), which are
+// site-wide rather than per-channel.
+async function treeUrls(
+ base: string,
+ tree: ArchiveTree,
+ slug: string | undefined,
+): Promise<string[]> {
+ const manifest = manifestUrl(tree, slug, base);
+ const m = await readManifest(manifest);
+ if (!m) return [];
+ const urls = [manifest];
+ for (let p = 0; p < (m.pageCount ?? 0); p++) {
+ urls.push(pageUrl(tree, slug, p, base));
+ }
+ return urls;
+}
+
export type DownloadProgress = { done: number; total: number };
-// Download every shard of a channel into the SW cache, reporting progress. The
-// URL list is the channel manifest plus each page-NNNN.json, prefixed with the
-// channel's origin so a hub can cache cross-origin channels.
+// Download every shard a channel needs into the SW cache, reporting progress.
+//
+// EVERY LAYER THAT HAS A MANIFEST, plus the root documents — not just
+// /transcripts. The list used to be transcripts-only, which is why the
+// duplicates page was dead offline (the report is a root file nothing
+// downloaded) and why an offline channel had no live chat, no posts and no AI
+// digests. Enumerating CONTRACT's trees means the next layer added to the
+// contract is offline-capable the day it ships, rather than the day someone
+// notices.
+//
+// Trees a site does not ship cost one 404 apiece and contribute nothing; the
+// root files are requested unconditionally because /duplicates.json and
+// /search-aliases.json are legitimately absent and the SW skips what it cannot
+// fetch. Transcripts are the one REQUIRED tree: no manifest there means the
+// channel is not readable and the download is refused, exactly as before.
export async function downloadChannelOffline(
origin: string,
slug: string,
@@ -83,13 +122,18 @@ export async function downloadChannelOffline(
): Promise<boolean> {
const sw = await controller();
if (!sw) return false;
- const manifest = await fetchManifest(origin, slug);
- if (!manifest) return false;
const base = idBaseUrl(origin);
- const urls = [`${base}/transcripts/${slug}/manifest.json`];
- for (let p = 0; p < manifest.pageCount; p++) {
- urls.push(`${base}/transcripts/${slug}/${pageFileName(p)}`);
+
+ const transcripts = await treeUrls(base, "transcripts", slug);
+ if (transcripts.length === 0) return false;
+
+ const urls = [...transcripts];
+ for (const tree of ARCHIVE_TREES) {
+ if (tree === "transcripts") continue;
+ urls.push(...(await treeUrls(base, tree, isFlatTree(tree) ? undefined : slug)));
}
+ for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base));
+
await sendToSw(sw, { type: "CACHE_URLS", origin, slug, urls }, (data) => {
if (data.type === "progress") {
onProgress?.({ done: data.done, total: data.total });
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -8,12 +8,18 @@
* Caches:
* SHELL — app shell: hashed /_next/static/* (immutable, cache-first) + HTML
* navigations (network-first, cache fallback) + icons/manifest.
- * PAGES — transcript/subs/summaries JSON shards (manifest.json + page-NNNN.json).
- * Pages are cache-first; manifests are network-first so a rebuild is
- * seen. Invalidation is keyed off manifest.generatedAt: when a channel's
- * manifest generatedAt changes, that channel's cached page shards are
- * evicted (the shards have stable, non-content-hashed URLs, so this is
- * the only reliable staleness signal).
+ * PAGES — every published JSON document: the per-channel shard trees
+ * (manifest.json + page-NNNN.json), the site-wide flat trees
+ * (summaries, stats) and the root documents (corpus.json, site.json,
+ * search-aliases.json, duplicates.json). Per-channel pages are
+ * cache-first; per-channel manifests are network-first so a rebuild is
+ * seen, with invalidation keyed off manifest.generatedAt — when a
+ * channel's manifest generatedAt changes, that channel's cached page
+ * shards are evicted (the shards have stable, non-content-hashed URLs,
+ * so this is the only reliable staleness signal). The site-wide
+ * documents have no per-channel generatedAt to evict against, so they
+ * are network-first with a cache fallback: fresh online, present
+ * offline, never stale-forever.
* META — tiny synthetic Responses storing each channel's last-seen generatedAt
* (avoids IndexedDB inside the SW).
*/
@@ -23,8 +29,18 @@ const SHELL = `shell-${VERSION}`;
const PAGES = `pages-${VERSION}`;
const META = `meta-${VERSION}`;
-// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees.
-const SHARD_RE = /^\/(transcripts|subs|posts|digests|summaries)\/([^/]+)\/(.+)$/;
+// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT. A service worker cannot
+// import, so these lists are hand-written — and `common/lib/archive/contract.
+// test.ts` reads THIS FILE and fails if they drift from CONTRACT.layers,
+// PER_CHANNEL_TREES and ROOT_FILES. Add a layer to the contract and this file
+// is what the test sends you to.
+
+// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json).
+const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
+// Flat trees: one manifest and its pages at the tree root, no channel level.
+const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
+// The root documents a reader fetches by name.
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
@@ -60,6 +76,16 @@ self.addEventListener("fetch", (event) => {
return;
}
+ // Site-wide archive documents. Not cache-first: there is no per-channel
+ // generatedAt to evict them against, so a cache-first copy would be stale
+ // until the SW version changed. Network-first keeps them fresh online and
+ // present offline — which is what made the duplicates page work with no
+ // connection.
+ if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) {
+ event.respondWith(networkFirst(req, PAGES));
+ return;
+ }
+
// App shell.
if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) {
event.respondWith(cacheFirst(req, SHELL));
@@ -86,6 +112,20 @@ async function cacheFirst(req, cacheName) {
}
}
+// Network-first: serve the fresh copy and remember it; fall back to whatever is
+// cached when the network is gone.
+async function networkFirst(req, cacheName) {
+ const cache = await caches.open(cacheName);
+ try {
+ const res = await fetch(req);
+ if (res.ok) cache.put(req, res.clone());
+ return res;
+ } catch (err) {
+ const hit = await cache.match(req);
+ return hit || Response.error();
+ }
+}
+
// Navigations: network-first (fresh HTML after deploys), fall back to cache, then
// to any cached document so the installed app still opens offline.
async function networkFirstDoc(req) {
@@ -135,10 +175,14 @@ async function handleManifest(req, slug) {
async function evictChannelPages(slug) {
const pages = await caches.open(PAGES);
const keys = await pages.keys();
+ // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed
+ // (summaries is flat) and /posts, /digests were missing, so a rebuilt channel
+ // kept serving its old social posts and AI digests from cache.
const prefixes = [
`/transcripts/${slug}/`,
`/subs/${slug}/`,
- `/summaries/${slug}/`,
+ `/posts/${slug}/`,
+ `/digests/${slug}/`,
];
await Promise.all(
keys.map((k) => {
diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js
@@ -2,7 +2,7 @@
*
* Same offline model as the site SW (browse-cache + opt-in per-channel
* download), but CROSS-ORIGIN: the hub federates content from many origins, so
- * this worker caches transcript/subs/summaries shards from ANY origin — not
+ * this worker caches every published archive document from ANY origin — not
* just its own. The only origins it ever reaches are ones the user added, whose
* JSON is CORS-open (Access-Control-Allow-Origin: *); a cross-origin fetch is
* therefore a readable `cors` response that can be cached (never `no-cors` —
@@ -21,9 +21,19 @@ const SHELL = `shell-${VERSION}`;
const PAGES = `pages-${VERSION}`;
const META = `meta-${VERSION}`;
-// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees, on
-// any origin (we test url.pathname, so it's origin-independent).
-const SHARD_RE = /^\/(transcripts|subs|posts|summaries)\/([^/]+)\/(.+)$/;
+// THE THREE URL FAMILIES OF THE PUBLISHED CONTRACT, matched on url.pathname so
+// they are origin-independent. A service worker cannot import, so these lists
+// are hand-written — and `common/lib/archive/contract.test.ts` reads THIS FILE
+// (and site-sw.js) and fails if they drift from CONTRACT.layers,
+// PER_CHANNEL_TREES and ROOT_FILES. `digests` was missing here entirely: the
+// hub federated AI digests it could never cache.
+
+// Per-channel trees: /<tree>/<slug>/(manifest.json|page-NNNN.json).
+const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
+// Flat trees: one manifest and its pages at the tree root, no channel level.
+const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
+// The root documents a reader fetches by name.
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
self.addEventListener("install", () => {
self.skipWaiting();
@@ -57,6 +67,15 @@ self.addEventListener("fetch", (event) => {
return;
}
+ // Site-wide archive documents, also from ANY origin — a member site's
+ // summaries index and its /duplicates.json are as federated as its shards.
+ // Network-first: no per-channel generatedAt to evict them against, so fresh
+ // online and present offline rather than stale-forever.
+ if (FLAT_RE.test(url.pathname) || ROOT_RE.test(url.pathname)) {
+ event.respondWith(networkFirst(req, PAGES));
+ return;
+ }
+
// App shell is same-origin only (the hub's own bundle).
if (url.origin !== self.location.origin) return;
if (url.pathname.startsWith("/_next/static/") || url.pathname.startsWith("/icons/")) {
@@ -83,6 +102,20 @@ async function cacheFirst(req, cacheName) {
}
}
+// Network-first: serve the fresh copy and remember it; fall back to whatever is
+// cached when the network is gone.
+async function networkFirst(req, cacheName) {
+ const cache = await caches.open(cacheName);
+ try {
+ const res = await fetch(req);
+ if (res.ok) cache.put(req, res.clone());
+ return res;
+ } catch (err) {
+ const hit = await cache.match(req);
+ return hit || Response.error();
+ }
+}
+
async function networkFirstDoc(req) {
const cache = await caches.open(SHELL);
try {
@@ -133,10 +166,13 @@ function belongsToChannel(entryUrl, origin, slug) {
const u = new URL(entryUrl);
const wantOrigin = origin || self.location.origin;
if (u.origin !== wantOrigin) return false;
+ // The PER-CHANNEL trees, and only those: /summaries/<slug>/ never existed
+ // (summaries is flat) and /posts, /digests were missing.
const prefixes = [
`/transcripts/${slug}/`,
`/subs/${slug}/`,
- `/summaries/${slug}/`,
+ `/posts/${slug}/`,
+ `/digests/${slug}/`,
];
return prefixes.some(
(pre) => u.pathname.startsWith(pre) && !u.pathname.endsWith("manifest.json"),