Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 872c984c70320e396fa1e61c5e607ebc09cfb4f6
parent 5eb761be296b90659bf2693096d7b27b1a7d73d3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 12 Sep 2026 02:48:47 -0400

common: tests for the reader registry and for the caches as wrappers

Fifteen cases, and every one of them counts fetches, because what had to
survive collapsing eight walks into one is not the shape of the data — tsc
covers that — but the memo behaviour that makes the viewer usable. A hit
returns the SAME promise, so N components asking for one video do not each
resolve on a different tick; a miss costs one read of each document, not one
per caller and not one per record.

The cases that would have caught a plausible wrong answer:

- the other record of a page we already paid for is a hit (transcriptCache's
  opportunistic warming, which is what makes a one-channel search cheap);
- an unreachable subs manifest is fetched TWICE, i.e. not memoised as absent
  — this is the whole reason the raw reads exist beside the tolerant ones;
- a thread walks each page of its channel once however many times it is
  opened, which is the postsCache page memo the reader does not provide;
- a channel with no digests costs exactly one 404 however many of its videos
  are opened, and a video missing from slugToPage is null rather than a throw;
- the same-origin reader's URLs are asserted as a literal list, root-relative
  and byte-identical to what the caches built before — the export e2e's shard
  route-mocks depend on that and would otherwise fail far from the cause.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/components/archiveCaches.test.ts | 303+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive/readers.test.ts | 181+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 484 insertions(+), 0 deletions(-)

diff --git a/common/components/archiveCaches.test.ts b/common/components/archiveCaches.test.ts @@ -0,0 +1,303 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { resetReaders } from "../lib/archive/readers"; +import { fetchTranscript } from "./transcriptCache"; +import { fetchSubs } from "./subsCache"; +import { fetchPost, fetchThread, peekPost } from "./postsCache"; +import { fetchDigest, hasDigest } from "./digestCache"; +import { fetchDuplicates, fetchDuplicateReport } from "./duplicatesCache"; +import { fetchAliases } from "./aliasesCache"; + +// ─── the eight caches as WRAPPERS ─── +// +// Each of these used to hand-roll the manifest -> slugToPage -> page walk; each +// now calls the shared ArchiveReader. What must survive that move is the memo +// behaviour, because it is what makes the viewer usable: a hit returns the same +// promise (so N components asking for the same video do not each re-render on a +// different tick) and a miss costs ONE read of each document, not one per +// caller and not one per record. +// +// So every test below counts fetches. The caches hold module-level state with +// no reset hook — that is deliberate, it is a page-lifetime cache — so each +// test uses its own channel slug and its own origin rather than trying to wipe +// them. + +type Archive = { fetches: string[]; body: Map<string, unknown> }; + +function installFetch(entries: Record<string, unknown>): { + archive: Archive; + restore: () => void; +} { + const archive: Archive = { + fetches: [], + body: new Map(Object.entries(entries)), + }; + const real = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + archive.fetches.push(url); + if (!archive.body.has(url)) { + return new Response("not found", { status: 404, statusText: "Not Found" }); + } + return new Response(JSON.stringify(archive.body.get(url)), { status: 200 }); + }) as typeof fetch; + return { + archive, + restore: () => { + globalThis.fetch = real; + resetReaders(); + }, + }; +} + +function count(archive: Archive, url: string): number { + return archive.fetches.filter((u) => u === url).length; +} + +test("transcriptCache: one manifest + one page, and the whole page is warmed", async () => { + const { archive, restore } = installFetch({ + "/transcripts/tc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + generatedAt: "", + }, + "/transcripts/tc/page-0000.json": [ + { slug: "tc/a", id: "a", title: "A", cues: [] }, + { slug: "tc/b", id: "b", title: "B", cues: [] }, + ], + }); + try { + // Two concurrent callers for the same id share one promise — not two reads + // that happen to agree. + const p1 = fetchTranscript("tc/a"); + const p2 = fetchTranscript("tc/a"); + assert.equal(p1, p2); + assert.equal((await p1).title, "A"); + + // A hit resolves without touching the network at all. + const after = archive.fetches.length; + assert.equal((await fetchTranscript("tc/a")).title, "A"); + assert.equal(archive.fetches.length, after); + + // And the OTHER record of the page we already paid for is a hit too: the + // opportunistic warming is what makes a search over one channel cheap. + assert.equal((await fetchTranscript("tc/b")).title, "B"); + assert.equal(count(archive, "/transcripts/tc/manifest.json"), 1); + assert.equal(count(archive, "/transcripts/tc/page-0000.json"), 1); + assert.equal(archive.fetches.length, 2); + } finally { + restore(); + } +}); + +test("transcriptCache: a video the manifest does not list never costs a page read", async () => { + const { archive, restore } = installFetch({ + "/transcripts/tcmiss/manifest.json": { + version: 1, + slugToPage: {}, + pageCount: 0, + generatedAt: "", + }, + }); + try { + await assert.rejects(() => fetchTranscript("tcmiss/nope"), /Unknown transcript slug/); + assert.deepEqual(archive.fetches, ["/transcripts/tcmiss/manifest.json"]); + } finally { + restore(); + } +}); + +test("subsCache: same walk, same sharing, over the subs tree", async () => { + const { archive, restore } = installFetch({ + "/subs/sc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + generatedAt: "", + }, + "/subs/sc/page-0000.json": [ + { slug: "sc/a", id: "a", tracks: {} }, + { slug: "sc/b", id: "b", tracks: {} }, + ], + }); + try { + const p1 = fetchSubs("sc/a"); + assert.equal(fetchSubs("sc/a"), p1); + assert.equal((await p1).id, "a"); + assert.equal((await fetchSubs("sc/b")).id, "b"); + assert.deepEqual(archive.fetches, [ + "/subs/sc/manifest.json", + "/subs/sc/page-0000.json", + ]); + } finally { + restore(); + } +}); + +test("subsCache: an unreachable channel manifest is NOT memoised as absent", async () => { + const { archive, restore } = installFetch({}); + try { + await assert.rejects(() => fetchSubs("scfail/a")); + await assert.rejects(() => fetchSubs("scfail/a")); + // Twice, because a failure is a failure and not an answer. The reader's own + // tolerant subsManifest() would have cached `null` here and the panel would + // stay empty for the session. + assert.equal(count(archive, "/subs/scfail/manifest.json"), 2); + } finally { + restore(); + } +}); + +test("postsCache: a thread walks every page of the channel exactly once", async () => { + const { archive, restore } = installFetch({ + "/posts/pc/manifest.json": { + version: 1, + slugToPage: { p1: 0, p2: 1 }, + pageCount: 2, + generatedAt: "", + }, + "/posts/pc/page-0000.json": [ + { slug: "pc/p1", id: "p1", threadId: "p1", createdAt: "2026-01-01" }, + ], + "/posts/pc/page-0001.json": [ + { slug: "pc/p2", id: "p2", threadId: "p1", createdAt: "2026-01-02" }, + ], + }); + try { + assert.equal((await fetchPost("pc/p1")).id, "p1"); + // Fetching the post warmed the page it came from, so peekPost is sync. + assert.equal(peekPost("pc/p1")?.id, "p1"); + + const thread = await fetchThread("pc/p1"); + assert.deepEqual( + thread.map((t) => t.id), + ["p1", "p2"], + ); + // A second thread read is free: the page memo is what stops fetchThread + // re-downloading the whole channel per post opened. + await fetchThread("pc/p1"); + assert.equal(count(archive, "/posts/pc/manifest.json"), 1); + assert.equal(count(archive, "/posts/pc/page-0000.json"), 1); + assert.equal(count(archive, "/posts/pc/page-0001.json"), 1); + } finally { + restore(); + } +}); + +test("digestCache: a channel with no digests costs ONE 404, ever", async () => { + const { archive, restore } = installFetch({}); + try { + assert.equal(await hasDigest("dcnone/a"), false); + assert.equal(await fetchDigest("dcnone/a"), null); + assert.equal(await fetchDigest("dcnone/b"), null); + // The sparse layer's whole cost model: absence is an answer and is cached, + // so opening video after video in an undigested channel is free. + assert.equal(count(archive, "/digests/dcnone/manifest.json"), 1); + } finally { + restore(); + } +}); + +test("digestCache: present, page-warmed, and not-in-slugToPage means null not throw", async () => { + const { archive, restore } = installFetch({ + "/digests/dc/manifest.json": { + version: 1, + slugToPage: { a: 0, b: 0 }, + pageCount: 1, + pageHashes: ["h0"], + generatedAt: "", + }, + "/digests/dc/page-0000.json": [ + { slug: "dc/a", id: "a", chapters: [] }, + { slug: "dc/b", id: "b", chapters: [] }, + ], + }); + try { + assert.equal(await hasDigest("dc/a"), true); + assert.equal((await fetchDigest("dc/a"))?.id, "a"); + assert.equal((await fetchDigest("dc/b"))?.id, "b"); + // An undigested video in a digested channel: normal, and not an error. + assert.equal(await hasDigest("dc/c"), false); + assert.equal(await fetchDigest("dc/c"), null); + assert.equal(count(archive, "/digests/dc/manifest.json"), 1); + assert.equal(count(archive, "/digests/dc/page-0000.json"), 1); + } finally { + restore(); + } +}); + +test("duplicatesCache: absent is empty, present is a slug-keyed lookup", async () => { + const absent = installFetch({}); + try { + assert.equal(await fetchDuplicateReport("https://dup-absent.example"), null); + assert.equal((await fetchDuplicates("https://dup-absent.example")).size, 0); + } finally { + absent.restore(); + } + + const O = "https://dup-present.example"; + const present = installFetch({ + [`${O}/duplicates.json`]: { + clusters: [ + { + clusterId: "c1", + videoRefs: [{ slug: "x/1" }, { slug: "y/2" }], + }, + ], + }, + }); + try { + const lookup = await fetchDuplicates(O); + assert.equal(lookup.size, 2); + assert.equal(lookup.get("x/1")?.clusterId, "c1"); + assert.equal(lookup.get("y/2")?.clusterId, "c1"); + assert.equal(count(present.archive, `${O}/duplicates.json`), 1); + } finally { + present.restore(); + } +}); + +test("aliasesCache: a 404 is an empty dictionary; a transport failure is rethrown", async () => { + const missing = installFetch({}); + try { + assert.deepEqual(await fetchAliases("https://alias-404.example"), []); + } finally { + missing.restore(); + } + + const present = installFetch({ + "https://alias-ok.example/search-aliases.json": { + aliases: [ + { + id: "kcup", + label: "k cups", + triggers: ["k cups"], + suggestion: "(k|cake)[ -]?cup", + useRegex: true, + }, + ], + }, + }); + try { + const aliases = await fetchAliases("https://alias-ok.example"); + assert.equal(aliases.length, 1); + assert.equal(aliases[0].id, "kcup"); + } finally { + present.restore(); + } + + // The distinction that matters: the query holding this has staleTime + // Infinity, so swallowing a blip would mean "this site has no aliases" for + // the rest of the session. + const real = globalThis.fetch; + globalThis.fetch = (async () => { + throw new TypeError("network down"); + }) as typeof fetch; + try { + await assert.rejects(() => fetchAliases("https://alias-down.example"), /network down/); + } finally { + globalThis.fetch = real; + resetReaders(); + } +}); diff --git a/common/lib/archive/readers.test.ts b/common/lib/archive/readers.test.ts @@ -0,0 +1,181 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { ArchiveHttpError, RemoteSource } from "./reader"; +import { channelRef, readerFor, resetReaders } from "./readers"; + +// The same in-memory-archive trick as reader.test.ts: a Map from URL to JSON +// body installed as `fetch`, so every assertion below is about the URLs the +// reader actually asks for. The viewer's caches are thin wrappers over these +// reads, so pinning the reads pins the wire. + +type Archive = { fetches: string[]; body: Map<string, unknown> }; + +function installFetch(entries: Record<string, unknown>): { + archive: Archive; + restore: () => void; +} { + const archive: Archive = { + fetches: [], + body: new Map(Object.entries(entries)), + }; + const real = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(input); + archive.fetches.push(url); + if (!archive.body.has(url)) { + return new Response("not found", { status: 404, statusText: "Not Found" }); + } + return new Response(JSON.stringify(archive.body.get(url)), { status: 200 }); + }) as typeof fetch; + return { + archive, + restore: () => { + globalThis.fetch = real; + resetReaders(); + }, + }; +} + +test("readerFor: one reader per origin, kept for the life of the page", () => { + resetReaders(); + const same = readerFor(""); + assert.equal(readerFor(""), same); + assert.equal(readerFor(), same, "the default argument is the same-origin key"); + const other = readerFor("https://b.example"); + assert.notEqual(other, same); + assert.equal(readerFor("https://b.example"), other); + assert.ok(same instanceof RemoteSource); + resetReaders(); + assert.notEqual(readerFor(""), same, "resetReaders drops the caches with it"); +}); + +test("the same-origin reader emits ROOT-RELATIVE urls, byte-identical to the old walk", async () => { + const { archive, restore } = installFetch({ + "/summaries/manifest.json": { version: 3, pageCount: 2, channels: [] }, + "/summaries/page-0001.json": [], + "/stats/manifest.json": { version: 1, pageCount: 1, channels: [] }, + "/stats/page-0000.json": [], + "/subs/manifest.json": { version: 4, channels: [] }, + "/posts/manifest.json": { version: 1, channels: [] }, + "/search-aliases.json": { aliases: [] }, + "/duplicates.json": { clusters: [] }, + "/transcripts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + "/subs/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + "/posts/alpha/manifest.json": { slugToPage: {}, pageCount: 1 }, + }); + try { + const r = readerFor(""); + await r.readSummariesManifest(); + await r.readSummariesPage(1); + await r.readStatsManifest(); + await r.readStatsPage(0); + await r.readSubsSiteManifest(); + await r.readPostsSiteManifest(); + await r.readAliasConfig(); + await r.readDuplicates(); + await r.readChannelSubsManifest("alpha"); + await r.readChannelPostsManifest("alpha"); + assert.deepEqual(archive.fetches, [ + "/summaries/manifest.json", + "/summaries/page-0001.json", + "/stats/manifest.json", + "/stats/page-0000.json", + "/subs/manifest.json", + "/posts/manifest.json", + "/search-aliases.json", + "/duplicates.json", + "/subs/alpha/manifest.json", + "/posts/alpha/manifest.json", + ]); + } finally { + restore(); + } +}); + +test("a federated origin prefixes every one of those, and nothing else changes", async () => { + const O = "https://member.example"; + const { archive, restore } = installFetch({ + [`${O}/summaries/manifest.json`]: { version: 3, pageCount: 0, channels: [] }, + [`${O}/duplicates.json`]: { clusters: [] }, + }); + try { + const r = readerFor(O); + await r.readSummariesManifest(); + await r.readDuplicates(); + assert.deepEqual(archive.fetches, [ + `${O}/summaries/manifest.json`, + `${O}/duplicates.json`, + ]); + } finally { + restore(); + } +}); + +test("a raw read rejects with the status; the tolerant fold over it resolves empty", async () => { + const { restore } = installFetch({}); + try { + const r = readerFor(""); + await assert.rejects( + () => r.readDuplicates(), + (err: unknown) => + err instanceof ArchiveHttpError && + err.status === 404 && + /GET \/duplicates\.json -> 404/.test((err as Error).message), + ); + // Same document, same 404, through the fold the MCP uses. + assert.equal((await r.duplicateIndex()).size, 0); + // And through the summaries fold, which a partial/absent index must not + // turn into an error — the page planner reads empty as "scan anyway". + assert.equal((await r.videoIndex()).size, 0); + assert.deepEqual(await r.loadAliases(), []); + } finally { + restore(); + } +}); + +test("a digests manifest 404 is the layer's own answer; any other status is not", async () => { + const { archive, restore } = installFetch({ + "/digests/beta/manifest.json": { slugToPage: { v1: 0 }, pageCount: 1 }, + }); + try { + const r = readerFor(""); + assert.equal(await r.readChannelDigestsManifest("alpha"), null); + assert.ok(await r.readChannelDigestsManifest("beta")); + assert.deepEqual(archive.fetches, [ + "/digests/alpha/manifest.json", + "/digests/beta/manifest.json", + ]); + } finally { + restore(); + } + + const bad = installFetch({}); + const realFetch = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + void input; + return new Response("boom", { status: 500, statusText: "Server Error" }); + }) as typeof fetch; + try { + await assert.rejects( + () => readerFor("").readChannelDigestsManifest("alpha"), + /-> 500/, + ); + } finally { + globalThis.fetch = realFetch; + bad.restore(); + } +}); + +test("channelRef: slug-addressed, and carries the origin only when there is one", () => { + assert.deepEqual(channelRef("alpha"), { + key: "alpha", + slug: "alpha", + name: "alpha", + }); + assert.deepEqual(channelRef("alpha", "https://b.example"), { + key: "alpha", + slug: "alpha", + name: "alpha", + siteUrl: "https://b.example", + }); +});