Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit ba3ac4b279eae097e4b5bb9c07c79d81eb0fc9d2
parent 93155dc703fcb4ba5caf216148fdfa0b06ab96c9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 13:48:21 -0400

common: videoTitles — a channel's video titles from index, scan store, metadata.info.json

readChannelVideoTitles merges the three sources cheapest first (one
byChannel key range + sums gets, one metadata-scan.json read, a 16 KB head
read of metadata.info.json only for the remainder). readVideoMetadataForDisplay
is the video page's header: metadata.info.json, else the scan entry.
Measured ~82 ms for a synthetic 5,000-id channel.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/videoTitles.test.ts | 257+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/videoTitles.ts | 241+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 498 insertions(+), 0 deletions(-)

diff --git a/common/controller/videoTitles.test.ts b/common/controller/videoTitles.test.ts @@ -0,0 +1,257 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, writeFile, readdir } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { upsertMetadataScan } from "./metadataScanStore"; +import { + readChannelVideoTitles, + readVideoMetadataForDisplay, +} from "./videoTitles"; + +// Run with: pnpm -C common exec tsx --test "controller/videoTitles.test.ts" +// +// A REAL LMDB file written the way buildIndex writes it (compression on, one +// fat description) — the reader must open it with compression too, and a fake +// would not catch that (curatedTagsPreview.test.ts's lesson). + +const SLUG = "ch"; + +async function fixture(prefix: string): Promise<{ dir: string; paths: Paths }> { + const dir = await mkdtemp(path.join(tmpdir(), prefix)); + const channelsDir = path.join(dir, "channels"); + await mkdir(path.join(channelsDir, SLUG, "data"), { recursive: true }); + return { + dir, + paths: { lmdbPath: path.join(dir, "index.mdb"), channelsDir } as Paths, + }; +} + +function summary(id: string, title: string, uploadDate: string): TranscriptSummary { + return { + slug: `${SLUG}/${id}`, + id, + channelSlug: SLUG, + title, + uploadDate, + duration: 60, + channel: "Ch", + // Over lmdb-js's ~1 KB compression threshold. + description: "d".repeat(4000), + tags: [], + isLivestream: false, + ageRestricted: false, + platform: "youtube", + webpageUrl: `https://www.youtube.com/watch?v=${id}`, + } as TranscriptSummary; +} + +async function writeIndex( + lmdbPath: string, + rows: ReadonlyArray<{ slug: string; summary: TranscriptSummary }>, +): Promise<void> { + const { open } = await import("lmdb"); + const root = open({ path: lmdbPath, maxDbs: 18, compression: true }); + const sums = root.openDB<TranscriptSummary, [string, string, string]>({ + name: "sums", + encoding: "msgpack", + }); + const byChannel = root.openDB<number, [string, string, string]>({ + name: "byChannel", + encoding: "msgpack", + }); + for (const { slug, summary: s } of rows) { + await sums.put([s.uploadDate, slug, s.id], s); + await byChannel.put([slug, s.uploadDate, s.id], 1); + } + await sums.flushed; + await byChannel.flushed; + await root.close(); +} + +async function writeInfo(paths: Paths, id: string, info: object): Promise<void> { + const dir = path.join(paths.channelsDir, SLUG, "data", id); + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info)); +} + +async function scan( + paths: Paths, + entries: Record<string, { title: string; description?: string; duration?: number }>, +): Promise<void> { + const full: Parameters<typeof upsertMetadataScan>[2]["entries"] = {}; + for (const [id, e] of Object.entries(entries)) { + full[id] = { + title: e.title, + description: e.description ?? "", + uploadDate: "20260101", + duration: e.duration, + scannedAt: "2026-09-25T00:00:00.000Z", + }; + } + await upsertMetadataScan(paths, SLUG, { entries: full }, "2026-09-25T00:00:00.000Z"); +} + +test("merges index, scan and metadata in that order; first hit wins; bare ids are absent", async () => { + const { dir, paths } = await fixture("vtitles-merge-"); + try { + await writeIndex(paths.lmdbPath, [ + { slug: SLUG, summary: summary("idx", "From the index", "20260101") }, + { slug: SLUG, summary: summary("both", "Index beats scan", "20260102") }, + // Another channel's id must not leak into this one. + { slug: "other", summary: summary("foreign", "Other channel", "20260103") }, + ]); + await scan(paths, { + both: { title: "Scan loses" }, + scanned: { title: "From the scan" }, + }); + await writeInfo(paths, "disk", { id: "disk", title: "From metadata.info.json" }); + await writeInfo(paths, "idx", { id: "idx", title: "Metadata loses" }); + + const got = await readChannelVideoTitles(paths, SLUG, [ + "idx", + "both", + "scanned", + "disk", + "bare", + "foreign", + ]); + assert.deepEqual(Object.fromEntries(got), { + idx: { title: "From the index", source: "index" }, + both: { title: "Index beats scan", source: "index" }, + scanned: { title: "From the scan", source: "scan" }, + disk: { title: "From metadata.info.json", source: "metadata" }, + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a missing index and a missing scan store are not errors", async () => { + const { dir, paths } = await fixture("vtitles-missing-"); + try { + await writeInfo(paths, "a", { id: "a", title: "Only on disk" }); + const got = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.deepEqual(Object.fromEntries(got), { + a: { title: "Only on disk", source: "metadata" }, + }); + assert.equal((await readChannelVideoTitles(paths, SLUG, [])).size, 0); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("an index title equal to the id (summarize's fallback) is not a title", async () => { + const { dir, paths } = await fixture("vtitles-fallback-"); + try { + await writeIndex(paths.lmdbPath, [ + { slug: SLUG, summary: summary("noname", "noname", "20260101") }, + ]); + await scan(paths, { noname: { title: "Named by the scan" } }); + const got = await readChannelVideoTitles(paths, SLUG, ["noname"]); + assert.deepEqual(got.get("noname"), { + title: "Named by the scan", + source: "scan", + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("metadata head read: title past the head and escaped titles still resolve", async () => { + const { dir, paths } = await fixture("vtitles-head-"); + try { + // A title that JSON must unescape. + await writeInfo(paths, "esc", { id: "esc", title: 'He said "hi" \\ bye ✓' }); + // A title beyond the 16 KB head: the full-parse fallback finds it. + await writeInfo(paths, "late", { id: "late", pad: "x".repeat(40_000), title: "Late title" }); + // The scan never creates a video dir, so reading must not either. + const got = await readChannelVideoTitles(paths, SLUG, ["esc", "late", "ghost"]); + assert.equal(got.get("esc")?.title, 'He said "hi" \\ bye ✓'); + assert.equal(got.get("late")?.title, "Late title"); + assert.equal(got.has("ghost"), false); + const dirs = await readdir(path.join(paths.channelsDir, SLUG, "data")); + assert.deepEqual(dirs.sort(), ["esc", "late"]); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("readVideoMetadataForDisplay: metadata.info.json, else the scan entry, else none", async () => { + const { dir, paths } = await fixture("vtitles-display-"); + try { + await writeInfo(paths, "dl", { + id: "dl", + title: "Downloaded", + description: "Full description", + webpage_url: "https://www.youtube.com/watch?v=dl", + uploader: "Uploader", + upload_date: "20250102", + duration: 125, + }); + await scan(paths, { + dl: { title: "Scan must lose" }, + un: { title: "Listed only", description: "Scan description", duration: 61 }, + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "dl"), { + title: "Downloaded", + description: "Full description", + webpageUrl: "https://www.youtube.com/watch?v=dl", + uploader: "Uploader", + uploadDate: "20250102", + duration: 125, + source: "metadata", + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "un"), { + title: "Listed only", + description: "Scan description", + uploadDate: "20260101", + duration: 61, + source: "scan", + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "nope"), { + source: "none", + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +// The cost bar: a 5,000-id channel, all three sources exercised. Recorded in +// plans/release-8.md; asserted only loosely so a slow CI box does not flake. +test("cost: 5,000 ids across the three sources", async () => { + const { dir, paths } = await fixture("vtitles-cost-"); + try { + const ids = Array.from({ length: 5000 }, (_, i) => `v${String(i).padStart(5, "0")}`); + // 3,000 in the index, 1,500 in the scan, 400 on disk only, 100 bare. + await writeIndex( + paths.lmdbPath, + ids.slice(0, 3000).map((id, i) => ({ + slug: SLUG, + summary: summary(id, `Indexed ${id}`, `2025${String((i % 12) + 1).padStart(2, "0")}01`), + })), + ); + await scan( + paths, + Object.fromEntries( + ids.slice(3000, 4500).map((id) => [id, { title: `Scanned ${id}`, description: "x".repeat(500) }]), + ), + ); + for (const id of ids.slice(4500, 4900)) { + await writeInfo(paths, id, { id, title: `Disk ${id}`, formats: "f".repeat(50_000) }); + } + const t0 = performance.now(); + const got = await readChannelVideoTitles(paths, SLUG, ids); + const ms = performance.now() - t0; + assert.equal(got.size, 4900); + const bySource = { index: 0, scan: 0, metadata: 0 }; + for (const v of got.values()) bySource[v.source]++; + assert.deepEqual(bySource, { index: 3000, scan: 1500, metadata: 400 }); + console.log(`readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms`); + assert.ok(ms < 5000, `took ${ms} ms`); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/common/controller/videoTitles.ts b/common/controller/videoTitles.ts @@ -0,0 +1,241 @@ +// What a channel's videos are CALLED, for the editor's per-channel video list +// and the video page — without downloading anything and without walking the +// corpus. +// +// Three sources, cheapest first, first hit wins: +// +// 1. "index" — the LMDB `sums` sub-DB (TranscriptSummary.title), read by a +// key range over `byChannel` for this one channel. Covers the +// videos the index admitted, i.e. those with a +// metadata.info.json on disk at the last build. +// 2. "scan" — channels/<slug>/metadata-scan.json (metadataScanStore.ts): +// listed-but-undownloaded videos, only after a metadata scan. +// ONE file read for the whole channel. +// 3. "metadata" — data/<id>/metadata.info.json, for the remainder only +// (downloaded after the last index build, or a directory +// name that is not the index's metadata id — see below). A +// head read per id, not a full parse: some of these files +// are hundreds of KB. +// +// THE INDEX IS KEYED BY METADATA ID, THE LIST BY DIRECTORY NAME. On most +// channels they are the same string; on Rumble the directory is the URL slug +// and the metadata id is the embed id (recencyIndex.ts's layer-1 comment). A +// miss there is not wrong, it just falls through to source 3 and costs a file +// read. Nothing is matched fuzzily, so a title is never attributed to the +// wrong video. +// +// Read-only throughout: the index is opened `readOnly`, the scan store through +// its own loader, and no video directory is created (the scan store's +// invariant). + +import { existsSync } from "node:fs"; +import { open as openFile } from "node:fs/promises"; +import path from "node:path"; +import { open } from "lmdb"; +import type { Paths } from "../lib/paths"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { loadRawMetadataFromDir } from "../lib/transcripts-server"; +import { mapConcurrent } from "../lib/concurrency"; +import { loadMetadataScan } from "./metadataScanStore"; + +export type VideoTitleSource = "index" | "scan" | "metadata"; + +export type VideoTitle = { title: string; source: VideoTitleSource }; + +// buildIndex.ts's key shapes: sums is [uploadDate, slug, id], byChannel is +// [slug, uploadDate, id]. +type IndexKey = [string, string, string]; +type ChannelKey = [string, string, string]; + +// yt-dlp writes `"id"` then `"title"` first in metadata.info.json, so 16 KB of +// head finds the top-level title without reading the formats/subtitles tail. +const HEAD_BYTES = 16384; +const TITLE_RE = /"title":\s*("(?:[^"\\]|\\.)*")/; +const METADATA_READ_CONCURRENCY = 16; + +function channelDataDir(paths: Paths, slug: string): string { + return path.join(paths.channelsDir, slug, "data"); +} + +// Source 1. Never throws: a missing, locked or mid-rebuild index is "no +// titles from the index", and the other two sources carry on. +function readIndexTitles( + paths: Paths, + slug: string, + wanted: ReadonlySet<string>, + out: Map<string, VideoTitle>, +): void { + if (wanted.size === 0 || !existsSync(paths.lmdbPath)) return; + let root: ReturnType<typeof open>; + try { + // `compression: true` is not optional on a reader: buildIndex writes with + // it, and without it every value over ~1 KB (any real description) throws + // on decode. See curatedTagsPreview.ts's openIndex. + root = open({ + path: paths.lmdbPath, + readOnly: true, + maxDbs: 18, + compression: true, + }); + } catch { + return; + } + try { + const sums = root.openDB<TranscriptSummary, IndexKey>({ + name: "sums", + encoding: "msgpack", + }); + const byChannel = root.openDB<number, ChannelKey>({ + name: "byChannel", + encoding: "msgpack", + }); + // Key-only walk of this channel's range; a summary is decoded only for an + // id the caller asked about. + for (const { key } of byChannel.getRange({ + start: [slug], + end: [slug, "￿"], + })) { + const ck = key as ChannelKey; + if (ck[0] !== slug) break; + const id = ck[2]; + if (!wanted.has(id) || out.has(id)) continue; + const summary = sums.get([ck[1], ck[0], id]); + const title = summary?.title; + // summarize() falls back to the id when the metadata had no title — that + // is not a title, so leave the id for a later source. + if (typeof title === "string" && title.trim() && title !== id) { + out.set(id, { title, source: "index" }); + } + } + } catch { + // A partial read is still useful; whatever landed in `out` stands. + } finally { + void root.close().catch(() => {}); + } +} + +// Source 3, one id. A head read and a regex; a full parse only when the head +// did not contain a title (an unusual key order, or a pretty-printer that put +// it later). +async function readMetadataTitle(videoDir: string): Promise<string | null> { + const file = path.join(videoDir, "metadata.info.json"); + let head: string; + try { + const fh = await openFile(file, "r"); + try { + const buf = Buffer.alloc(HEAD_BYTES); + const { bytesRead } = await fh.read(buf, 0, HEAD_BYTES, 0); + head = buf.subarray(0, bytesRead).toString("utf8"); + } finally { + await fh.close(); + } + } catch { + return null; + } + const m = TITLE_RE.exec(head); + if (m) { + try { + const t = JSON.parse(m[1]) as unknown; + if (typeof t === "string" && t.trim()) return t; + } catch { + // fall through to the full parse + } + } + const meta = await loadRawMetadataFromDir(videoDir); + return typeof meta?.title === "string" && meta.title.trim() + ? meta.title + : null; +} + +// Titles for `ids` of one channel. An id with no title from any source is +// absent from the map — the caller shows the id. +export async function readChannelVideoTitles( + paths: Paths, + channelSlug: string, + ids: readonly string[], +): Promise<Map<string, VideoTitle>> { + const out = new Map<string, VideoTitle>(); + if (ids.length === 0) return out; + const wanted = new Set(ids); + + readIndexTitles(paths, channelSlug, wanted, out); + + if (out.size < wanted.size) { + const scan = await loadMetadataScan(paths, channelSlug); + for (const id of wanted) { + if (out.has(id)) continue; + const title = scan.entries[id]?.title; + if (title && title.trim()) out.set(id, { title, source: "scan" }); + } + } + + const remainder = [...wanted].filter((id) => !out.has(id)); + if (remainder.length > 0) { + const dataDir = channelDataDir(paths, channelSlug); + const titles = await mapConcurrent( + remainder, + METADATA_READ_CONCURRENCY, + (id) => readMetadataTitle(path.join(dataDir, id)), + ); + remainder.forEach((id, i) => { + const title = titles[i]; + if (title) out.set(id, { title, source: "metadata" }); + }); + } + return out; +} + +export type VideoDisplayMetadata = { + title?: string; + description?: string; + webpageUrl?: string; + uploader?: string; + uploadDate?: string; + duration?: number; + // "metadata" = data/<id>/metadata.info.json; "scan" = the channel's + // metadata-scan.json entry (a listed video that was never downloaded); + // "none" = neither exists, and the page shows the bare id. + source: "metadata" | "scan" | "none"; +}; + +function str(v: unknown): string | undefined { + return typeof v === "string" && v !== "" ? v : undefined; +} + +// The video page's header. metadata.info.json first — it is what the download +// wrote and carries the uploader and URL — else the scan entry. +export async function readVideoMetadataForDisplay( + paths: Paths, + channelSlug: string, + id: string, +): Promise<VideoDisplayMetadata> { + const meta = await loadRawMetadataFromDir( + path.join(channelDataDir(paths, channelSlug), id), + ); + if (meta) { + return { + title: str(meta.title), + description: str(meta.description), + webpageUrl: str(meta.webpage_url), + uploader: str(meta.uploader), + uploadDate: str(meta.upload_date), + duration: + typeof meta.duration === "number" && Number.isFinite(meta.duration) + ? meta.duration + : undefined, + source: "metadata", + }; + } + const scan = await loadMetadataScan(paths, channelSlug); + const entry = scan.entries[id]; + if (entry) { + return { + title: str(entry.title), + description: str(entry.description), + uploadDate: str(entry.uploadDate), + duration: entry.duration, + source: "scan", + }; + } + return { source: "none" }; +}