Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d2235841bcda1a05aa73755bda39c1b51ba38c51
parent 7719bdcfb87fad3edb1c9d9412327ae3f457ca5b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 14:09:56 -0400

common: memoize metadata.info.json titles per channel, keyed by data/ mtime (review M1)

On a channel whose dir names never match the index (Rumble, legacy
YYYYMMDD_ dirs) every row fell through to a head read on every render —
the-quartering-rumble, 8,049 dirs: 5,331 ms cold, 5.5 ms memoized now. A new
or removed video dir drops the channel's entry; misses are not memoized; 64
channels max, LRU. The e2e invalidate-cache route clears it.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/videoTitles.test.ts | 59+++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/controller/videoTitles.ts | 85+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
Meditor/app/api/test/invalidate-cache/route.ts | 6++++++
3 files changed, 144 insertions(+), 6 deletions(-)

diff --git a/common/controller/videoTitles.test.ts b/common/controller/videoTitles.test.ts @@ -1,6 +1,6 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdir, mkdtemp, rm, writeFile, readdir } from "node:fs/promises"; +import { mkdir, mkdtemp, rm, unlink, utimes, writeFile, readdir } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; @@ -9,6 +9,8 @@ import { upsertMetadataScan } from "./metadataScanStore"; import { readChannelVideoTitles, readVideoMetadataForDisplay, + resetVideoTitleMemo, + videoTitleMetadataReadCount, } from "./videoTitles"; // Run with: pnpm -C common exec tsx --test "controller/videoTitles.test.ts" @@ -249,9 +251,62 @@ test("cost: 5,000 ids across the three sources", async () => { const bySource = { index: 0, scan: 0, metadata: 0 }; for (const v of got.values()) bySource[v.source]++; assert.deepEqual(bySource, { index: 3000, scan: 1500, metadata: 400 }); - console.log(`readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms`); + const t1 = performance.now(); + await readChannelVideoTitles(paths, SLUG, ids); + const memoMs = performance.now() - t1; + console.log( + `readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms first, ${memoMs.toFixed(1)} ms memoized`, + ); assert.ok(ms < 5000, `took ${ms} ms`); } finally { await rm(dir, { recursive: true, force: true }); } }); + +test("metadata titles are memoized per channel until data/ changes", async () => { + const { dir, paths } = await fixture("vtitles-memo-"); + try { + resetVideoTitleMemo(); + const dataDir = path.join(paths.channelsDir, SLUG, "data"); + await writeInfo(paths, "a", { id: "a", title: "Alpha" }); + await writeInfo(paths, "b", { id: "b", title: "Bravo" }); + // Pin the dir's mtime so the second call's key is certainly unchanged. + const pinned = new Date("2026-01-01T00:00:00Z"); + await utimes(dataDir, pinned, pinned); + + const r0 = videoTitleMetadataReadCount(); + const first = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.equal(first.get("a")?.title, "Alpha"); + assert.equal(videoTitleMetadataReadCount() - r0, 2); + + // Same data/ mtime: served from the memo, nothing read — even with the + // file gone (proof that it was not read). + await unlink(path.join(dataDir, "a", "metadata.info.json")); + await utimes(dataDir, pinned, pinned); + const r1 = videoTitleMetadataReadCount(); + const second = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.equal(second.get("a")?.title, "Alpha"); + assert.equal(second.get("b")?.title, "Bravo"); + assert.equal(videoTitleMetadataReadCount() - r1, 0); + + // A new video dir changes data/'s mtime: the channel's memo is dropped and + // every id is read again ("a" now has no file, so no title). + await writeInfo(paths, "c", { id: "c", title: "Charlie" }); + const r2 = videoTitleMetadataReadCount(); + const third = await readChannelVideoTitles(paths, SLUG, ["a", "b", "c"]); + assert.equal(third.has("a"), false); + assert.equal(third.get("c")?.title, "Charlie"); + assert.equal(videoTitleMetadataReadCount() - r2, 3); + + // A miss is not memoized: a title that lands later is found. + await writeFile( + path.join(dataDir, "a", "metadata.info.json"), + JSON.stringify({ id: "a", title: "Alpha again" }), + ); + const fourth = await readChannelVideoTitles(paths, SLUG, ["a"]); + assert.equal(fourth.get("a")?.title, "Alpha again"); + } finally { + resetVideoTitleMemo(); + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/common/controller/videoTitles.ts b/common/controller/videoTitles.ts @@ -29,7 +29,7 @@ // invariant). import { existsSync } from "node:fs"; -import { open as openFile } from "node:fs/promises"; +import { open as openFile, stat } from "node:fs/promises"; import path from "node:path"; import { open } from "lmdb"; import type { Paths } from "../lib/paths"; @@ -147,6 +147,69 @@ async function readMetadataTitle(videoDir: string): Promise<string | null> { : null; } +// --- The source-3 memo -------------------------------------------------------- +// +// WHY: on a channel whose directory names never match the index (Rumble: dir = +// URL slug, index key = embed id; legacy `YYYYMMDD_<id>` dirs) EVERY row falls +// through to a metadata.info.json read, and the list page re-renders on every +// row click (force-dynamic, rows are `?video=` links). the-quartering-rumble has +// 8,049 such dirs: 5,024 ms cold / 224 ms warm per render, measured 2026-09-25. +// +// Per channel, keyed by the `data/` directory's mtime: adding or removing a +// video dir changes it and drops the channel's entry. Rewriting a +// metadata.info.json INSIDE an existing dir does not — accepted, because a +// video's title does not change after download. Misses (no file yet, e.g. a dir +// mid-download) are NOT memoized, so a title that lands later is picked up. +// Held on globalThis because Next can load this module more than once. +const MEMO_MAX_CHANNELS = 64; + +type TitleMemoEntry = { dataDirMtimeMs: number; titles: Map<string, string> }; + +declare global { + var __yttVideoTitleMemo__: Map<string, TitleMemoEntry> | undefined; +} + +function titleMemo(): Map<string, TitleMemoEntry> { + return (globalThis.__yttVideoTitleMemo__ ??= new Map()); +} + +// For tests and the e2e reset (api/test/invalidate-cache). +export function resetVideoTitleMemo(): void { + globalThis.__yttVideoTitleMemo__ = undefined; +} + +// Test hook: metadata.info.json reads so far in this process (tests diff it). +let metadataReads = 0; +export function videoTitleMetadataReadCount(): number { + return metadataReads; +} + +async function memoForChannel( + key: string, + dataDir: string, +): Promise<Map<string, string> | null> { + let mtimeMs: number; + try { + mtimeMs = (await stat(dataDir)).mtimeMs; + } catch { + return null; // no data/ — nothing to read, nothing to memoize + } + const memo = titleMemo(); + let entry = memo.get(key); + if (!entry || entry.dataDirMtimeMs !== mtimeMs) { + entry = { dataDirMtimeMs: mtimeMs, titles: new Map() }; + } + // Re-insert so iteration order is least-recently-used first. + memo.delete(key); + memo.set(key, entry); + while (memo.size > MEMO_MAX_CHANNELS) { + const oldest = memo.keys().next().value; + if (oldest === undefined) break; + memo.delete(oldest); + } + return entry.titles; +} + // Titles for `ids` of one channel. An id with no title from any source is // absent from the map — the caller shows the id. export async function readChannelVideoTitles( @@ -169,17 +232,31 @@ export async function readChannelVideoTitles( } } - const remainder = [...wanted].filter((id) => !out.has(id)); + let remainder = [...wanted].filter((id) => !out.has(id)); if (remainder.length > 0) { const dataDir = channelDataDir(paths, channelSlug); + const memo = await memoForChannel(dataDir, dataDir); + if (!memo) return out; + remainder = remainder.filter((id) => { + const title = memo.get(id); + if (title === undefined) return true; + out.set(id, { title, source: "metadata" }); + return false; + }); const titles = await mapConcurrent( remainder, METADATA_READ_CONCURRENCY, - (id) => readMetadataTitle(path.join(dataDir, id)), + (id) => { + metadataReads++; + return readMetadataTitle(path.join(dataDir, id)); + }, ); remainder.forEach((id, i) => { const title = titles[i]; - if (title) out.set(id, { title, source: "metadata" }); + if (title) { + memo.set(id, title); + out.set(id, { title, source: "metadata" }); + } }); } return out; diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts @@ -3,6 +3,7 @@ import { revalidatePath } from "next/cache"; import { resetSnapshotScheduler } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { resetChannelSnapshotMemo } from "yt-dlp-transcript-common/controller/channels"; import { resetStorageProbeMemo } from "yt-dlp-transcript-common/controller/storageLocations"; +import { resetVideoTitleMemo } from "yt-dlp-transcript-common/controller/videoTitles"; import { testRouteDenied } from "../_guard"; export const dynamic = "force-dynamic"; @@ -105,6 +106,11 @@ function invalidate() { // testInfo.outputPath varies by accident rather than on purpose; cleared here // so it is on purpose. resetStorageProbeMemo(); + // And the video list's metadata.info.json title memo. It is keyed by the + // channel's data/ mtime, which resetData() changes by recreating the dir — but + // a spec that rewrites a title in place inside one mtime tick would otherwise + // see the previous spec's title. + resetVideoTitleMemo(); revalidatePath("/", "layout"); return NextResponse.json({ ok: true }); }