Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 528d6d82c24ff2d0a947ac79d9315a01123dae55
parent 531f72173b1e4958845f2ccd45cdd9f28d8707d4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 11:57:41 -0400

export: an offline copy is two lists — per channel, and once per origin

Review defect. The first cut appended every ARCHIVE_TREES entry plus all four
ROOT_FILES to EACH channel's download list, which put the site-wide documents
in a per-channel bill. Measured on jeralyzer that is summaries ~14 MB + stats
~21.6 MB (page-0000 alone 20.97 MB) + duplicates.json 5.9 MB = ~41.5 MB of
byte-identical data per channel pinned, and re-fetched for real because the
worker bulk-caches with `cache: "reload"`.

Worse than the download: it could never be removed. Both evict paths sweep
/<tree>/<slug>/ prefixes, and none of those bytes live under a channel prefix,
so "remove offline copy" left all 41.5 MB in PAGES forever with no way to get
it back out.

So the list splits in two, in common/lib/archive/offlineUrls.ts: the
per-channel trees of one channel, and the flat trees plus root files of one
origin. Site data rides along with the FIRST channel pinned on an origin and
is skipped after that; a new EVICT_SITE message removes it when the last
pinned channel of that origin goes. Downloaded once, evicted once. The
per-channel download is byte-for-byte what it was.

Whether the site data is already present is answered from the pinned registry
rather than a new status round-trip — one pinned channel means one downloaded
copy, because the two are written and removed together. That inherits the
registry's existing drift (a viewer who clears site data while keeping
localStorage), which is the same gap channelCachedPages already exists for and
which a re-pin repairs.

The builders live in common/ rather than export/ for a reason beyond tidiness:
export has no test runner, and this split is exactly the thing that needs a
test. Six of them, asserting the channel list contains no flat tree and no
root file, that every channel URL is under a prefix the workers can actually
evict, that the two lists are disjoint and together cover every tree the
contract defines, and that an absent tree costs one probe.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/lib/archive/offlineUrls.test.ts | 178+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive/offlineUrls.ts | 102+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/offlineCache.ts | 101++++++++++++++++++++++++++++++++++++++-----------------------------------------
Mexport/service-worker/site-sw.js | 20++++++++++++++++++++
Mexport/service-worker/sw-hub.js | 24++++++++++++++++++++++++
5 files changed, 373 insertions(+), 52 deletions(-)

diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts @@ -0,0 +1,178 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { ARCHIVE_TREES, PER_CHANNEL_TREES, ROOT_FILES, isFlatTree } from "./contract"; +import { + channelArchiveUrls, + siteArchiveUrls, + type ManifestReader, +} from "./offlineUrls"; + +// THE SPLIT IS THE ASSERTION. An offline copy is two lists, and if a site-wide +// document ever leaks back into the per-channel one it is not a style problem: +// the per-channel list is paid per channel pinned, and the per-channel EVICT +// sweep (a /<tree>/<slug>/ prefix match in both service workers) cannot remove +// anything that is not under a channel prefix. So a flat-tree URL in the +// channel list is bytes that are downloaded N times and never removable. +// +// Measured on jeralyzer when that was the shipped behaviour: summaries ~14 MB + +// stats ~21.6 MB + duplicates 5.9 MB = ~41.5 MB per channel, re-fetched for +// real (the worker bulk-caches with `cache: "reload"`). + +// A stub archive: manifests present at these URLs with these page counts, +// everything else absent. Records the reads so "absent costs one probe" is +// assertable. +function reader(pageCounts: Record<string, number>): { + read: ManifestReader; + reads: string[]; +} { + const reads: string[] = []; + const read: ManifestReader = async (url) => { + reads.push(url); + return url in pageCounts ? { pageCount: pageCounts[url] } : null; + }; + return { read, reads }; +} + +const FULL = { + "/transcripts/alpha/manifest.json": 2, + "/subs/alpha/manifest.json": 1, + "/posts/alpha/manifest.json": 1, + "/digests/alpha/manifest.json": 1, + "/summaries/manifest.json": 2, + "/stats/manifest.json": 1, +}; + +test("the channel list is the PER-CHANNEL trees and nothing else", async () => { + const { read } = reader(FULL); + const urls = await channelArchiveUrls(read, "", "alpha"); + + assert.deepEqual(urls, [ + "/transcripts/alpha/manifest.json", + "/transcripts/alpha/page-0000.json", + "/transcripts/alpha/page-0001.json", + "/subs/alpha/manifest.json", + "/subs/alpha/page-0000.json", + "/posts/alpha/manifest.json", + "/posts/alpha/page-0000.json", + "/digests/alpha/manifest.json", + "/digests/alpha/page-0000.json", + ]); + + // Stated structurally as well as literally, so adding a layer to the contract + // fails here rather than silently shipping a channel that is half offline. + const trees = new Set(urls.map((u) => u.split("/")[1])); + assert.deepEqual([...trees].sort(), [...PER_CHANNEL_TREES].sort()); + + // Not one site-wide byte. This is the regression the review caught. + for (const u of urls) { + const tree = u.split("/")[1]; + assert.ok( + !ARCHIVE_TREES.some((t) => t === tree && isFlatTree(t)), + `flat tree ${tree} must not be in a per-channel download: ${u}`, + ); + } + for (const file of ROOT_FILES) { + assert.ok(!urls.includes(`/${file}`), `${file} must not be per channel`); + } + // Every URL is under a /<tree>/<slug>/ prefix — the only shape the service + // workers' per-channel evict sweep can ever remove. + for (const u of urls) { + assert.match(u, /^\/[^/]+\/alpha\//, `${u} is not evictable per channel`); + } +}); + +test("the site list is the FLAT trees plus the root files, and nothing per-channel", async () => { + const { read } = reader(FULL); + const urls = await siteArchiveUrls(read, ""); + + assert.deepEqual(urls, [ + "/summaries/manifest.json", + "/summaries/page-0000.json", + "/summaries/page-0001.json", + "/stats/manifest.json", + "/stats/page-0000.json", + "/corpus.json", + "/site.json", + "/search-aliases.json", + "/duplicates.json", + ]); + + const flat = ARCHIVE_TREES.filter(isFlatTree); + for (const tree of flat) { + assert.ok( + urls.some((u) => u.startsWith(`/${tree}/`)), + `${tree} missing from the site list`, + ); + } + for (const file of ROOT_FILES) assert.ok(urls.includes(`/${file}`)); + // No channel slug anywhere: nothing here is paid per channel. + for (const u of urls) assert.doesNotMatch(u, /alpha/); +}); + +test("together the two lists cover every tree the contract defines, once", async () => { + const { read } = reader(FULL); + const all = [ + ...(await channelArchiveUrls(read, "", "alpha")), + ...(await siteArchiveUrls(read, "")), + ]; + assert.equal(new Set(all).size, all.length, "no URL is in both lists"); + + const trees = new Set( + all.filter((u) => u.split("/").length > 2).map((u) => u.split("/")[1]), + ); + assert.deepEqual([...trees].sort(), [...ARCHIVE_TREES].sort()); +}); + +test("a tree the site does not ship costs one probe and contributes nothing", async () => { + // Transcripts only — the shape of a site with no chat, no posts, no digests + // and no stats. + const { read, reads } = reader({ "/transcripts/alpha/manifest.json": 1 }); + + assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), [ + "/transcripts/alpha/manifest.json", + "/transcripts/alpha/page-0000.json", + ]); + // One probe per absent tree, not a retry storm. + assert.deepEqual(reads, [ + "/transcripts/alpha/manifest.json", + "/subs/alpha/manifest.json", + "/posts/alpha/manifest.json", + "/digests/alpha/manifest.json", + ]); + + // The root files are listed unprobed — absent ones are skipped by the worker. + assert.deepEqual(await siteArchiveUrls(read, ""), [ + "/corpus.json", + "/site.json", + "/search-aliases.json", + "/duplicates.json", + ]); +}); + +test("no transcripts manifest means no download at all", async () => { + const { read } = reader({ "/subs/alpha/manifest.json": 1 }); + // A channel whose transcripts are unreachable cannot be opened offline, so + // caching its live chat would be caching a dead end. + assert.deepEqual(await channelArchiveUrls(read, "", "alpha"), []); +}); + +test("a federated origin prefixes both lists and nothing else changes", async () => { + const O = "https://member.example"; + const { read } = reader({ + [`${O}/transcripts/alpha/manifest.json`]: 1, + [`${O}/summaries/manifest.json`]: 1, + }); + + assert.deepEqual(await channelArchiveUrls(read, O, "alpha"), [ + `${O}/transcripts/alpha/manifest.json`, + `${O}/transcripts/alpha/page-0000.json`, + ]); + assert.deepEqual(await siteArchiveUrls(read, O), [ + `${O}/summaries/manifest.json`, + `${O}/summaries/page-0000.json`, + `${O}/corpus.json`, + `${O}/site.json`, + `${O}/search-aliases.json`, + `${O}/duplicates.json`, + ]); +}); diff --git a/common/lib/archive/offlineUrls.ts b/common/lib/archive/offlineUrls.ts @@ -0,0 +1,102 @@ +// WHAT AN OFFLINE COPY OF AN ARCHIVE CONSISTS OF, split the way it is paid for. +// +// Two lists, and the split is the whole point of this module: +// +// channelArchiveUrls — the PER-CHANNEL trees of one channel. Pinning a second +// channel genuinely costs this again, because none of it +// is shared. +// siteArchiveUrls — the SITE-WIDE documents of one origin: the flat trees +// (summaries, stats) and the root files. Identical for +// every channel of that origin, so it is downloaded once +// and evicted when the last pinned channel goes. +// +// The first cut of this shipped as ONE list and it was a real cost, not a +// tidiness point. Measured on jeralyzer: summaries ~14 MB, stats ~21.6 MB (its +// page-0000 alone is 20.97 MB) and /duplicates.json 5.9 MB — ~41.5 MB of +// byte-identical site data re-fetched per channel pinned (the service worker +// bulk-fetches with `cache: "reload"`, so they really do go over the wire), and +// the per-channel evict path sweeps only the per-channel prefixes, so removing +// a channel left every byte of it behind forever. +// +// The manifest READER IS INJECTED. The caller owns the transport, which matters +// here: the viewer reads these with `cache: "no-store"` precisely because a +// normal fetch would be answered from the service-worker cache this list exists +// to refill, and would then enumerate the stale copy. Injection also makes the +// lists directly assertable without a network. + +import { + ARCHIVE_TREES, + PER_CHANNEL_TREES, + ROOT_FILES, + isFlatTree, + manifestUrl, + pageUrl, + rootFileUrl, +} from "./contract"; + +// Every manifest shape these walks need, reduced to the one field a URL list is +// built from. A tree a site does not ship answers 404 and reads as absent. +export type PagedManifest = { pageCount?: number }; + +// Reads one manifest by URL. Resolves null for "this site does not ship it", +// which for these lists is normal rather than an error. +export type ManifestReader = (url: string) => Promise<PagedManifest | null>; + +// The manifest plus every page of one tree, or [] when the tree is absent. +// `slug` is undefined for a flat tree. +async function treeUrls( + read: ManifestReader, + base: string, + tree: (typeof ARCHIVE_TREES)[number], + slug: string | undefined, +): Promise<string[]> { + const manifest = manifestUrl(tree, slug, base); + const m = await read(manifest); + if (!m) return []; + const urls = [manifest]; + for (let p = 0; p < (m.pageCount ?? 0); p++) { + urls.push(pageUrl(tree, slug, p, base)); + } + return urls; +} + +// One channel's shards, across every per-channel tree the contract defines. +// +// Returns [] when the channel has no TRANSCRIPTS manifest — that is not a tree +// a readable channel can be missing, so the caller refuses the download rather +// than caching a channel that cannot be opened. Every other tree is optional +// and costs one 404 when absent. +export async function channelArchiveUrls( + read: ManifestReader, + base: string, + slug: string, +): Promise<string[]> { + const transcripts = await treeUrls(read, base, "transcripts", slug); + if (transcripts.length === 0) return []; + const urls = [...transcripts]; + for (const tree of PER_CHANNEL_TREES) { + if (tree === "transcripts") continue; + urls.push(...(await treeUrls(read, base, tree, slug))); + } + return urls; +} + +// One origin's site-wide documents: the flat trees and the root files. +// +// The root files are listed unconditionally rather than probed. /duplicates.json +// and /search-aliases.json are legitimately absent on many sites, the caller +// hands the list to a worker that skips what it cannot fetch, and probing them +// first would double the request count to learn something the fetch already +// tells us. +export async function siteArchiveUrls( + read: ManifestReader, + base: string, +): Promise<string[]> { + const urls: string[] = []; + for (const tree of ARCHIVE_TREES) { + if (!isFlatTree(tree)) continue; + urls.push(...(await treeUrls(read, base, tree, undefined))); + } + for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base)); + return urls; +} diff --git a/export/app/lib/offlineCache.ts b/export/app/lib/offlineCache.ts @@ -12,15 +12,11 @@ // per (origin, slug); the site SW ignores it (same-origin only). import { - ARCHIVE_TREES, - ROOT_FILES, - isFlatTree, - manifestUrl, - pageUrl, - rootFileUrl, - type ArchiveTree, -} from "yt-dlp-transcript-common/lib/archive/contract"; -import { idBaseUrl, makeId } from "yt-dlp-transcript-common/components/originId"; + channelArchiveUrls, + siteArchiveUrls, + type PagedManifest, +} from "yt-dlp-transcript-common/lib/archive/offlineUrls"; +import { idBaseUrl, makeId, splitId } from "yt-dlp-transcript-common/components/originId"; const PINNED_KEY = "ytdlp-tb:offline-channels"; @@ -61,15 +57,12 @@ function sendToSw( }); } -// Every manifest shape the walk needs, reduced to the one field a URL list is -// built from. A tree a site does not ship answers 404 and reads as absent. -type PagedManifest = { pageCount?: number }; - // Deliberately NOT the shared ArchiveReader: the reader is a normal `fetch`, so // a page walking it would be answered from the very service-worker cache this // function exists to REFILL, and would compute its download list from the stale // copy. `cache: "no-store"` is the whole point of this read. What is shared is -// the thing that actually drifted — the URL shape. +// the thing that actually drifted — the URL shape, and the two lists in +// common/lib/archive/offlineUrls.ts. async function readManifest(url: string): Promise<PagedManifest | null> { try { const res = await fetch(url, { cache: "no-store" }); @@ -80,41 +73,36 @@ async function readManifest(url: string): Promise<PagedManifest | null> { } } -// The manifest + every page of one tree, or [] when the site ships no such -// tree. `slug` is undefined for the flat trees (summaries, stats), which are -// site-wide rather than per-channel. -async function treeUrls( - base: string, - tree: ArchiveTree, - slug: string | undefined, -): Promise<string[]> { - const manifest = manifestUrl(tree, slug, base); - const m = await readManifest(manifest); - if (!m) return []; - const urls = [manifest]; - for (let p = 0; p < (m.pageCount ?? 0); p++) { - urls.push(pageUrl(tree, slug, p, base)); - } - return urls; -} - export type DownloadProgress = { done: number; total: number }; -// Download every shard a channel needs into the SW cache, reporting progress. +// Is any channel of this origin already pinned? That is the test for whether +// the origin's SITE DATA (summaries, stats, the root files) is already in the +// cache, and it is deliberately answered from the pinned registry rather than +// by asking the worker: the two lists are written and evicted together, so one +// pinned channel means one downloaded copy of the site data. // -// EVERY LAYER THAT HAS A MANIFEST, plus the root documents — not just -// /transcripts. The list used to be transcripts-only, which is why the -// duplicates page was dead offline (the report is a root file nothing -// downloaded) and why an offline channel had no live chat, no posts and no AI -// digests. Enumerating CONTRACT's trees means the next layer added to the -// contract is offline-capable the day it ships, rather than the day someone -// notices. +// The failure mode this accepts is the one the registry already has — a viewer +// who clears site data while keeping localStorage gets a pin with no bytes +// behind it, which is exactly why channelCachedPages() exists. The honest fix +// is a status round-trip per origin; it is not worth a second SW message for a +// case a re-pin already repairs. +function originHasPin(origin: string): boolean { + return pinnedChannels().some((id) => splitId(id).origin === origin); +} + +// Download a channel for offline use, reporting progress. // -// Trees a site does not ship cost one 404 apiece and contribute nothing; the -// root files are requested unconditionally because /duplicates.json and -// /search-aliases.json are legitimately absent and the SW skips what it cannot -// fetch. Transcripts are the one REQUIRED tree: no manifest there means the -// channel is not readable and the download is refused, exactly as before. +// EVERY PER-CHANNEL LAYER THAT HAS A MANIFEST, not just /transcripts — that is +// why an offline channel now has its live chat, its social posts and its AI +// digests — plus, ONCE PER ORIGIN, the site-wide documents the viewer needs to +// render any of it: the summaries index, stats, and the root files. The +// duplicates report is one of those root files, which is why the duplicates page +// used to be dead with no connection. +// +// The site data is downloaded with the FIRST channel pinned on an origin and +// skipped for every channel after it. Pinning it per channel is ~41.5 MB of +// identical bytes each time on a corpus the size of jeralyzer, re-fetched for +// real because the worker bulk-caches with `cache: "reload"`. export async function downloadChannelOffline( origin: string, slug: string, @@ -124,15 +112,16 @@ export async function downloadChannelOffline( if (!sw) return false; const base = idBaseUrl(origin); - const transcripts = await treeUrls(base, "transcripts", slug); - if (transcripts.length === 0) return false; + const urls = await channelArchiveUrls(readManifest, base, slug); + // No transcripts manifest: the channel is not readable, so there is nothing + // to take offline. Refused before anything is written, exactly as before. + if (urls.length === 0) return false; - const urls = [...transcripts]; - for (const tree of ARCHIVE_TREES) { - if (tree === "transcripts") continue; - urls.push(...(await treeUrls(base, tree, isFlatTree(tree) ? undefined : slug))); + // Checked BEFORE this channel is pinned, so the first channel of an origin + // brings the site data with it and later ones do not. + if (!originHasPin(origin)) { + urls.push(...(await siteArchiveUrls(readManifest, base))); } - for (const file of ROOT_FILES) urls.push(rootFileUrl(file, base)); await sendToSw(sw, { type: "CACHE_URLS", origin, slug, urls }, (data) => { if (data.type === "progress") { @@ -143,6 +132,11 @@ export async function downloadChannelOffline( return true; } +// Remove a channel's offline copy — and, when it was the LAST pinned channel of +// its origin, that origin's site data with it. Without the second half the +// summaries/stats/root bytes survive every removal and there is no way to get +// them back out of the cache at all; they are downloaded once, so they have to +// be evicted once. export async function evictChannelOffline( origin: string, slug: string, @@ -152,6 +146,9 @@ export async function evictChannelOffline( await sendToSw(sw, { type: "EVICT_CHANNEL", origin, slug }).catch(() => {}); } unpin(origin, slug); + if (sw && !originHasPin(origin)) { + await sendToSw(sw, { type: "EVICT_SITE", origin }).catch(() => {}); + } } // How many page shards of a channel are currently cached (0 = not offline). diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js @@ -195,6 +195,22 @@ async function evictChannelPages(slug) { ); } +// The site-wide documents: the flat trees and the root files, i.e. exactly what +// the page downloads ONCE per origin. Evicted when its last pinned channel goes, +// because nothing else ever removes them — the per-channel sweep above only +// knows about /<tree>/<slug>/ prefixes. The same two regexes drive the fetch +// handler, so this can never fall out of step with what was cached. +async function evictSiteData() { + const pages = await caches.open(PAGES); + const keys = await pages.keys(); + await Promise.all( + keys.map((k) => { + const p = new URL(k.url).pathname; + return FLAT_RE.test(p) || ROOT_RE.test(p) ? pages.delete(k) : null; + }), + ); +} + // Per-channel generatedAt stored as a synthetic Response in the META cache. async function readMeta(slug) { const meta = await caches.open(META); @@ -216,6 +232,10 @@ self.addEventListener("message", (event) => { event.waitUntil( evictChannelPages(data.slug).then(() => port && port.postMessage({ ok: true })), ); + } else if (data.type === "EVICT_SITE") { + event.waitUntil( + evictSiteData().then(() => port && port.postMessage({ ok: true })), + ); } else if (data.type === "CHANNEL_STATUS") { event.waitUntil(channelStatus(data.slug, port)); } diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js @@ -187,6 +187,26 @@ async function evictChannelPages(origin, slug) { ); } +// One member origin's site-wide documents — the flat trees and the root files, +// exactly what the page downloads ONCE per origin. Scoped by origin like +// everything else here, so evicting one member never touches another's. The +// per-channel sweep above only knows /<tree>/<slug>/ prefixes, so without this +// these bytes would survive every removal. +async function evictSiteData(origin) { + const pages = await caches.open(PAGES); + const wantOrigin = origin || self.location.origin; + const keys = await pages.keys(); + await Promise.all( + keys.map((k) => { + const u = new URL(k.url); + if (u.origin !== wantOrigin) return null; + return FLAT_RE.test(u.pathname) || ROOT_RE.test(u.pathname) + ? pages.delete(k) + : null; + }), + ); +} + // Per-(origin, channel) generatedAt stored as a synthetic Response in META. function metaKey(origin, slug) { return `/__gen__/${encodeURIComponent(origin || "")}/${slug}`; @@ -214,6 +234,10 @@ self.addEventListener("message", (event) => { () => port && port.postMessage({ ok: true }), ), ); + } else if (data.type === "EVICT_SITE") { + event.waitUntil( + evictSiteData(origin).then(() => port && port.postMessage({ ok: true })), + ); } else if (data.type === "CHANNEL_STATUS") { event.waitUntil(channelStatus(origin, data.slug, port)); }