commit ba3ac4b279eae097e4b5bb9c07c79d81eb0fc9d2
parent 93155dc703fcb4ba5caf216148fdfa0b06ab96c9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 13:48:21 -0400
common: videoTitles — a channel's video titles from index, scan store, metadata.info.json
readChannelVideoTitles merges the three sources cheapest first (one
byChannel key range + sums gets, one metadata-scan.json read, a 16 KB head
read of metadata.info.json only for the remainder). readVideoMetadataForDisplay
is the video page's header: metadata.info.json, else the scan entry.
Measured ~82 ms for a synthetic 5,000-id channel.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 498 insertions(+), 0 deletions(-)
diff --git a/common/controller/videoTitles.test.ts b/common/controller/videoTitles.test.ts
@@ -0,0 +1,257 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, writeFile, readdir } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { TranscriptSummary } from "../lib/transcripts";
+import { upsertMetadataScan } from "./metadataScanStore";
+import {
+ readChannelVideoTitles,
+ readVideoMetadataForDisplay,
+} from "./videoTitles";
+
+// Run with: pnpm -C common exec tsx --test "controller/videoTitles.test.ts"
+//
+// A REAL LMDB file written the way buildIndex writes it (compression on, one
+// fat description) — the reader must open it with compression too, and a fake
+// would not catch that (curatedTagsPreview.test.ts's lesson).
+
+const SLUG = "ch";
+
+async function fixture(prefix: string): Promise<{ dir: string; paths: Paths }> {
+ const dir = await mkdtemp(path.join(tmpdir(), prefix));
+ const channelsDir = path.join(dir, "channels");
+ await mkdir(path.join(channelsDir, SLUG, "data"), { recursive: true });
+ return {
+ dir,
+ paths: { lmdbPath: path.join(dir, "index.mdb"), channelsDir } as Paths,
+ };
+}
+
+function summary(id: string, title: string, uploadDate: string): TranscriptSummary {
+ return {
+ slug: `${SLUG}/${id}`,
+ id,
+ channelSlug: SLUG,
+ title,
+ uploadDate,
+ duration: 60,
+ channel: "Ch",
+ // Over lmdb-js's ~1 KB compression threshold.
+ description: "d".repeat(4000),
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: `https://www.youtube.com/watch?v=${id}`,
+ } as TranscriptSummary;
+}
+
+async function writeIndex(
+ lmdbPath: string,
+ rows: ReadonlyArray<{ slug: string; summary: TranscriptSummary }>,
+): Promise<void> {
+ const { open } = await import("lmdb");
+ const root = open({ path: lmdbPath, maxDbs: 18, compression: true });
+ const sums = root.openDB<TranscriptSummary, [string, string, string]>({
+ name: "sums",
+ encoding: "msgpack",
+ });
+ const byChannel = root.openDB<number, [string, string, string]>({
+ name: "byChannel",
+ encoding: "msgpack",
+ });
+ for (const { slug, summary: s } of rows) {
+ await sums.put([s.uploadDate, slug, s.id], s);
+ await byChannel.put([slug, s.uploadDate, s.id], 1);
+ }
+ await sums.flushed;
+ await byChannel.flushed;
+ await root.close();
+}
+
+async function writeInfo(paths: Paths, id: string, info: object): Promise<void> {
+ const dir = path.join(paths.channelsDir, SLUG, "data", id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info));
+}
+
+async function scan(
+ paths: Paths,
+ entries: Record<string, { title: string; description?: string; duration?: number }>,
+): Promise<void> {
+ const full: Parameters<typeof upsertMetadataScan>[2]["entries"] = {};
+ for (const [id, e] of Object.entries(entries)) {
+ full[id] = {
+ title: e.title,
+ description: e.description ?? "",
+ uploadDate: "20260101",
+ duration: e.duration,
+ scannedAt: "2026-09-25T00:00:00.000Z",
+ };
+ }
+ await upsertMetadataScan(paths, SLUG, { entries: full }, "2026-09-25T00:00:00.000Z");
+}
+
+test("merges index, scan and metadata in that order; first hit wins; bare ids are absent", async () => {
+ const { dir, paths } = await fixture("vtitles-merge-");
+ try {
+ await writeIndex(paths.lmdbPath, [
+ { slug: SLUG, summary: summary("idx", "From the index", "20260101") },
+ { slug: SLUG, summary: summary("both", "Index beats scan", "20260102") },
+ // Another channel's id must not leak into this one.
+ { slug: "other", summary: summary("foreign", "Other channel", "20260103") },
+ ]);
+ await scan(paths, {
+ both: { title: "Scan loses" },
+ scanned: { title: "From the scan" },
+ });
+ await writeInfo(paths, "disk", { id: "disk", title: "From metadata.info.json" });
+ await writeInfo(paths, "idx", { id: "idx", title: "Metadata loses" });
+
+ const got = await readChannelVideoTitles(paths, SLUG, [
+ "idx",
+ "both",
+ "scanned",
+ "disk",
+ "bare",
+ "foreign",
+ ]);
+ assert.deepEqual(Object.fromEntries(got), {
+ idx: { title: "From the index", source: "index" },
+ both: { title: "Index beats scan", source: "index" },
+ scanned: { title: "From the scan", source: "scan" },
+ disk: { title: "From metadata.info.json", source: "metadata" },
+ });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a missing index and a missing scan store are not errors", async () => {
+ const { dir, paths } = await fixture("vtitles-missing-");
+ try {
+ await writeInfo(paths, "a", { id: "a", title: "Only on disk" });
+ const got = await readChannelVideoTitles(paths, SLUG, ["a", "b"]);
+ assert.deepEqual(Object.fromEntries(got), {
+ a: { title: "Only on disk", source: "metadata" },
+ });
+ assert.equal((await readChannelVideoTitles(paths, SLUG, [])).size, 0);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("an index title equal to the id (summarize's fallback) is not a title", async () => {
+ const { dir, paths } = await fixture("vtitles-fallback-");
+ try {
+ await writeIndex(paths.lmdbPath, [
+ { slug: SLUG, summary: summary("noname", "noname", "20260101") },
+ ]);
+ await scan(paths, { noname: { title: "Named by the scan" } });
+ const got = await readChannelVideoTitles(paths, SLUG, ["noname"]);
+ assert.deepEqual(got.get("noname"), {
+ title: "Named by the scan",
+ source: "scan",
+ });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("metadata head read: title past the head and escaped titles still resolve", async () => {
+ const { dir, paths } = await fixture("vtitles-head-");
+ try {
+ // A title that JSON must unescape.
+ await writeInfo(paths, "esc", { id: "esc", title: 'He said "hi" \\ bye ✓' });
+ // A title beyond the 16 KB head: the full-parse fallback finds it.
+ await writeInfo(paths, "late", { id: "late", pad: "x".repeat(40_000), title: "Late title" });
+ // The scan never creates a video dir, so reading must not either.
+ const got = await readChannelVideoTitles(paths, SLUG, ["esc", "late", "ghost"]);
+ assert.equal(got.get("esc")?.title, 'He said "hi" \\ bye ✓');
+ assert.equal(got.get("late")?.title, "Late title");
+ assert.equal(got.has("ghost"), false);
+ const dirs = await readdir(path.join(paths.channelsDir, SLUG, "data"));
+ assert.deepEqual(dirs.sort(), ["esc", "late"]);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("readVideoMetadataForDisplay: metadata.info.json, else the scan entry, else none", async () => {
+ const { dir, paths } = await fixture("vtitles-display-");
+ try {
+ await writeInfo(paths, "dl", {
+ id: "dl",
+ title: "Downloaded",
+ description: "Full description",
+ webpage_url: "https://www.youtube.com/watch?v=dl",
+ uploader: "Uploader",
+ upload_date: "20250102",
+ duration: 125,
+ });
+ await scan(paths, {
+ dl: { title: "Scan must lose" },
+ un: { title: "Listed only", description: "Scan description", duration: 61 },
+ });
+ assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "dl"), {
+ title: "Downloaded",
+ description: "Full description",
+ webpageUrl: "https://www.youtube.com/watch?v=dl",
+ uploader: "Uploader",
+ uploadDate: "20250102",
+ duration: 125,
+ source: "metadata",
+ });
+ assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "un"), {
+ title: "Listed only",
+ description: "Scan description",
+ uploadDate: "20260101",
+ duration: 61,
+ source: "scan",
+ });
+ assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "nope"), {
+ source: "none",
+ });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+// The cost bar: a 5,000-id channel, all three sources exercised. Recorded in
+// plans/release-8.md; asserted only loosely so a slow CI box does not flake.
+test("cost: 5,000 ids across the three sources", async () => {
+ const { dir, paths } = await fixture("vtitles-cost-");
+ try {
+ const ids = Array.from({ length: 5000 }, (_, i) => `v${String(i).padStart(5, "0")}`);
+ // 3,000 in the index, 1,500 in the scan, 400 on disk only, 100 bare.
+ await writeIndex(
+ paths.lmdbPath,
+ ids.slice(0, 3000).map((id, i) => ({
+ slug: SLUG,
+ summary: summary(id, `Indexed ${id}`, `2025${String((i % 12) + 1).padStart(2, "0")}01`),
+ })),
+ );
+ await scan(
+ paths,
+ Object.fromEntries(
+ ids.slice(3000, 4500).map((id) => [id, { title: `Scanned ${id}`, description: "x".repeat(500) }]),
+ ),
+ );
+ for (const id of ids.slice(4500, 4900)) {
+ await writeInfo(paths, id, { id, title: `Disk ${id}`, formats: "f".repeat(50_000) });
+ }
+ const t0 = performance.now();
+ const got = await readChannelVideoTitles(paths, SLUG, ids);
+ const ms = performance.now() - t0;
+ assert.equal(got.size, 4900);
+ const bySource = { index: 0, scan: 0, metadata: 0 };
+ for (const v of got.values()) bySource[v.source]++;
+ assert.deepEqual(bySource, { index: 3000, scan: 1500, metadata: 400 });
+ console.log(`readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms`);
+ assert.ok(ms < 5000, `took ${ms} ms`);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/controller/videoTitles.ts b/common/controller/videoTitles.ts
@@ -0,0 +1,241 @@
+// What a channel's videos are CALLED, for the editor's per-channel video list
+// and the video page — without downloading anything and without walking the
+// corpus.
+//
+// Three sources, cheapest first, first hit wins:
+//
+// 1. "index" — the LMDB `sums` sub-DB (TranscriptSummary.title), read by a
+// key range over `byChannel` for this one channel. Covers the
+// videos the index admitted, i.e. those with a
+// metadata.info.json on disk at the last build.
+// 2. "scan" — channels/<slug>/metadata-scan.json (metadataScanStore.ts):
+// listed-but-undownloaded videos, only after a metadata scan.
+// ONE file read for the whole channel.
+// 3. "metadata" — data/<id>/metadata.info.json, for the remainder only
+// (downloaded after the last index build, or a directory
+// name that is not the index's metadata id — see below). A
+// head read per id, not a full parse: some of these files
+// are hundreds of KB.
+//
+// THE INDEX IS KEYED BY METADATA ID, THE LIST BY DIRECTORY NAME. On most
+// channels they are the same string; on Rumble the directory is the URL slug
+// and the metadata id is the embed id (recencyIndex.ts's layer-1 comment). A
+// miss there is not wrong, it just falls through to source 3 and costs a file
+// read. Nothing is matched fuzzily, so a title is never attributed to the
+// wrong video.
+//
+// Read-only throughout: the index is opened `readOnly`, the scan store through
+// its own loader, and no video directory is created (the scan store's
+// invariant).
+
+import { existsSync } from "node:fs";
+import { open as openFile } from "node:fs/promises";
+import path from "node:path";
+import { open } from "lmdb";
+import type { Paths } from "../lib/paths";
+import type { TranscriptSummary } from "../lib/transcripts";
+import { loadRawMetadataFromDir } from "../lib/transcripts-server";
+import { mapConcurrent } from "../lib/concurrency";
+import { loadMetadataScan } from "./metadataScanStore";
+
+export type VideoTitleSource = "index" | "scan" | "metadata";
+
+export type VideoTitle = { title: string; source: VideoTitleSource };
+
+// buildIndex.ts's key shapes: sums is [uploadDate, slug, id], byChannel is
+// [slug, uploadDate, id].
+type IndexKey = [string, string, string];
+type ChannelKey = [string, string, string];
+
+// yt-dlp writes `"id"` then `"title"` first in metadata.info.json, so 16 KB of
+// head finds the top-level title without reading the formats/subtitles tail.
+const HEAD_BYTES = 16384;
+const TITLE_RE = /"title":\s*("(?:[^"\\]|\\.)*")/;
+const METADATA_READ_CONCURRENCY = 16;
+
+function channelDataDir(paths: Paths, slug: string): string {
+ return path.join(paths.channelsDir, slug, "data");
+}
+
+// Source 1. Never throws: a missing, locked or mid-rebuild index is "no
+// titles from the index", and the other two sources carry on.
+function readIndexTitles(
+ paths: Paths,
+ slug: string,
+ wanted: ReadonlySet<string>,
+ out: Map<string, VideoTitle>,
+): void {
+ if (wanted.size === 0 || !existsSync(paths.lmdbPath)) return;
+ let root: ReturnType<typeof open>;
+ try {
+ // `compression: true` is not optional on a reader: buildIndex writes with
+ // it, and without it every value over ~1 KB (any real description) throws
+ // on decode. See curatedTagsPreview.ts's openIndex.
+ root = open({
+ path: paths.lmdbPath,
+ readOnly: true,
+ maxDbs: 18,
+ compression: true,
+ });
+ } catch {
+ return;
+ }
+ try {
+ const sums = root.openDB<TranscriptSummary, IndexKey>({
+ name: "sums",
+ encoding: "msgpack",
+ });
+ const byChannel = root.openDB<number, ChannelKey>({
+ name: "byChannel",
+ encoding: "msgpack",
+ });
+ // Key-only walk of this channel's range; a summary is decoded only for an
+ // id the caller asked about.
+ for (const { key } of byChannel.getRange({
+ start: [slug],
+ end: [slug, ""],
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== slug) break;
+ const id = ck[2];
+ if (!wanted.has(id) || out.has(id)) continue;
+ const summary = sums.get([ck[1], ck[0], id]);
+ const title = summary?.title;
+ // summarize() falls back to the id when the metadata had no title — that
+ // is not a title, so leave the id for a later source.
+ if (typeof title === "string" && title.trim() && title !== id) {
+ out.set(id, { title, source: "index" });
+ }
+ }
+ } catch {
+ // A partial read is still useful; whatever landed in `out` stands.
+ } finally {
+ void root.close().catch(() => {});
+ }
+}
+
+// Source 3, one id. A head read and a regex; a full parse only when the head
+// did not contain a title (an unusual key order, or a pretty-printer that put
+// it later).
+async function readMetadataTitle(videoDir: string): Promise<string | null> {
+ const file = path.join(videoDir, "metadata.info.json");
+ let head: string;
+ try {
+ const fh = await openFile(file, "r");
+ try {
+ const buf = Buffer.alloc(HEAD_BYTES);
+ const { bytesRead } = await fh.read(buf, 0, HEAD_BYTES, 0);
+ head = buf.subarray(0, bytesRead).toString("utf8");
+ } finally {
+ await fh.close();
+ }
+ } catch {
+ return null;
+ }
+ const m = TITLE_RE.exec(head);
+ if (m) {
+ try {
+ const t = JSON.parse(m[1]) as unknown;
+ if (typeof t === "string" && t.trim()) return t;
+ } catch {
+ // fall through to the full parse
+ }
+ }
+ const meta = await loadRawMetadataFromDir(videoDir);
+ return typeof meta?.title === "string" && meta.title.trim()
+ ? meta.title
+ : null;
+}
+
+// Titles for `ids` of one channel. An id with no title from any source is
+// absent from the map — the caller shows the id.
+export async function readChannelVideoTitles(
+ paths: Paths,
+ channelSlug: string,
+ ids: readonly string[],
+): Promise<Map<string, VideoTitle>> {
+ const out = new Map<string, VideoTitle>();
+ if (ids.length === 0) return out;
+ const wanted = new Set(ids);
+
+ readIndexTitles(paths, channelSlug, wanted, out);
+
+ if (out.size < wanted.size) {
+ const scan = await loadMetadataScan(paths, channelSlug);
+ for (const id of wanted) {
+ if (out.has(id)) continue;
+ const title = scan.entries[id]?.title;
+ if (title && title.trim()) out.set(id, { title, source: "scan" });
+ }
+ }
+
+ const remainder = [...wanted].filter((id) => !out.has(id));
+ if (remainder.length > 0) {
+ const dataDir = channelDataDir(paths, channelSlug);
+ const titles = await mapConcurrent(
+ remainder,
+ METADATA_READ_CONCURRENCY,
+ (id) => readMetadataTitle(path.join(dataDir, id)),
+ );
+ remainder.forEach((id, i) => {
+ const title = titles[i];
+ if (title) out.set(id, { title, source: "metadata" });
+ });
+ }
+ return out;
+}
+
+export type VideoDisplayMetadata = {
+ title?: string;
+ description?: string;
+ webpageUrl?: string;
+ uploader?: string;
+ uploadDate?: string;
+ duration?: number;
+ // "metadata" = data/<id>/metadata.info.json; "scan" = the channel's
+ // metadata-scan.json entry (a listed video that was never downloaded);
+ // "none" = neither exists, and the page shows the bare id.
+ source: "metadata" | "scan" | "none";
+};
+
+function str(v: unknown): string | undefined {
+ return typeof v === "string" && v !== "" ? v : undefined;
+}
+
+// The video page's header. metadata.info.json first — it is what the download
+// wrote and carries the uploader and URL — else the scan entry.
+export async function readVideoMetadataForDisplay(
+ paths: Paths,
+ channelSlug: string,
+ id: string,
+): Promise<VideoDisplayMetadata> {
+ const meta = await loadRawMetadataFromDir(
+ path.join(channelDataDir(paths, channelSlug), id),
+ );
+ if (meta) {
+ return {
+ title: str(meta.title),
+ description: str(meta.description),
+ webpageUrl: str(meta.webpage_url),
+ uploader: str(meta.uploader),
+ uploadDate: str(meta.upload_date),
+ duration:
+ typeof meta.duration === "number" && Number.isFinite(meta.duration)
+ ? meta.duration
+ : undefined,
+ source: "metadata",
+ };
+ }
+ const scan = await loadMetadataScan(paths, channelSlug);
+ const entry = scan.entries[id];
+ if (entry) {
+ return {
+ title: str(entry.title),
+ description: str(entry.description),
+ uploadDate: str(entry.uploadDate),
+ duration: entry.duration,
+ source: "scan",
+ };
+ }
+ return { source: "none" };
+}