Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 83c81031e6af96e975ac099aa995feb97ad0821e
parent bdbaaa8995ce0ad349069ee551f9a201ed06b5f0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 14:17:33 -0400

Merge one-core/r8-video-titles — release 8 slice V: the channel video list shows and searches titles; the video page shows scan-store metadata for undownloaded videos

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/controller/videoTitles.test.ts | 312+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/videoTitles.ts | 318+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 2++
Meditor/app/api/test/invalidate-cache/route.ts | 6++++++
Meditor/app/channels/[slug]/components/VideoListPane.tsx | 30++++++++++++++++++++++++------
Aeditor/app/channels/[slug]/lib/videoRows.test.ts | 47+++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/lib/videoRows.ts | 15+++++++++++++++
Meditor/app/channels/[slug]/lib/videoRowsServer.ts | 5+++++
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 57+++++++++++++++++++++++++--------------------------------
Meditor/app/channels/[slug]/videos/page.tsx | 40++++++++++++++++++++--------------------
Aeditor/e2e/video-titles.spec.ts | 115+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aplans/release-8.md | 108+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
12 files changed, 997 insertions(+), 58 deletions(-)

diff --git a/common/controller/videoTitles.test.ts b/common/controller/videoTitles.test.ts @@ -0,0 +1,312 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, unlink, utimes, writeFile, readdir } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { upsertMetadataScan } from "./metadataScanStore"; +import { + readChannelVideoTitles, + readVideoMetadataForDisplay, + resetVideoTitleMemo, + videoTitleMetadataReadCount, +} from "./videoTitles"; + +// Run with: pnpm -C common exec tsx --test "controller/videoTitles.test.ts" +// +// A REAL LMDB file written the way buildIndex writes it (compression on, one +// fat description) — the reader must open it with compression too, and a fake +// would not catch that (curatedTagsPreview.test.ts's lesson). + +const SLUG = "ch"; + +async function fixture(prefix: string): Promise<{ dir: string; paths: Paths }> { + const dir = await mkdtemp(path.join(tmpdir(), prefix)); + const channelsDir = path.join(dir, "channels"); + await mkdir(path.join(channelsDir, SLUG, "data"), { recursive: true }); + return { + dir, + paths: { lmdbPath: path.join(dir, "index.mdb"), channelsDir } as Paths, + }; +} + +function summary(id: string, title: string, uploadDate: string): TranscriptSummary { + return { + slug: `${SLUG}/${id}`, + id, + channelSlug: SLUG, + title, + uploadDate, + duration: 60, + channel: "Ch", + // Over lmdb-js's ~1 KB compression threshold. + description: "d".repeat(4000), + tags: [], + isLivestream: false, + ageRestricted: false, + platform: "youtube", + webpageUrl: `https://www.youtube.com/watch?v=${id}`, + } as TranscriptSummary; +} + +async function writeIndex( + lmdbPath: string, + rows: ReadonlyArray<{ slug: string; summary: TranscriptSummary }>, +): Promise<void> { + const { open } = await import("lmdb"); + const root = open({ path: lmdbPath, maxDbs: 18, compression: true }); + const sums = root.openDB<TranscriptSummary, [string, string, string]>({ + name: "sums", + encoding: "msgpack", + }); + const byChannel = root.openDB<number, [string, string, string]>({ + name: "byChannel", + encoding: "msgpack", + }); + for (const { slug, summary: s } of rows) { + await sums.put([s.uploadDate, slug, s.id], s); + await byChannel.put([slug, s.uploadDate, s.id], 1); + } + await sums.flushed; + await byChannel.flushed; + await root.close(); +} + +async function writeInfo(paths: Paths, id: string, info: object): Promise<void> { + const dir = path.join(paths.channelsDir, SLUG, "data", id); + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify(info)); +} + +async function scan( + paths: Paths, + entries: Record<string, { title: string; description?: string; duration?: number }>, +): Promise<void> { + const full: Parameters<typeof upsertMetadataScan>[2]["entries"] = {}; + for (const [id, e] of Object.entries(entries)) { + full[id] = { + title: e.title, + description: e.description ?? "", + uploadDate: "20260101", + duration: e.duration, + scannedAt: "2026-09-25T00:00:00.000Z", + }; + } + await upsertMetadataScan(paths, SLUG, { entries: full }, "2026-09-25T00:00:00.000Z"); +} + +test("merges index, scan and metadata in that order; first hit wins; bare ids are absent", async () => { + const { dir, paths } = await fixture("vtitles-merge-"); + try { + await writeIndex(paths.lmdbPath, [ + { slug: SLUG, summary: summary("idx", "From the index", "20260101") }, + { slug: SLUG, summary: summary("both", "Index beats scan", "20260102") }, + // Another channel's id must not leak into this one. + { slug: "other", summary: summary("foreign", "Other channel", "20260103") }, + ]); + await scan(paths, { + both: { title: "Scan loses" }, + scanned: { title: "From the scan" }, + }); + await writeInfo(paths, "disk", { id: "disk", title: "From metadata.info.json" }); + await writeInfo(paths, "idx", { id: "idx", title: "Metadata loses" }); + + const got = await readChannelVideoTitles(paths, SLUG, [ + "idx", + "both", + "scanned", + "disk", + "bare", + "foreign", + ]); + assert.deepEqual(Object.fromEntries(got), { + idx: { title: "From the index", source: "index" }, + both: { title: "Index beats scan", source: "index" }, + scanned: { title: "From the scan", source: "scan" }, + disk: { title: "From metadata.info.json", source: "metadata" }, + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a missing index and a missing scan store are not errors", async () => { + const { dir, paths } = await fixture("vtitles-missing-"); + try { + await writeInfo(paths, "a", { id: "a", title: "Only on disk" }); + const got = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.deepEqual(Object.fromEntries(got), { + a: { title: "Only on disk", source: "metadata" }, + }); + assert.equal((await readChannelVideoTitles(paths, SLUG, [])).size, 0); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("an index title equal to the id (summarize's fallback) is not a title", async () => { + const { dir, paths } = await fixture("vtitles-fallback-"); + try { + await writeIndex(paths.lmdbPath, [ + { slug: SLUG, summary: summary("noname", "noname", "20260101") }, + ]); + await scan(paths, { noname: { title: "Named by the scan" } }); + const got = await readChannelVideoTitles(paths, SLUG, ["noname"]); + assert.deepEqual(got.get("noname"), { + title: "Named by the scan", + source: "scan", + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("metadata head read: title past the head and escaped titles still resolve", async () => { + const { dir, paths } = await fixture("vtitles-head-"); + try { + // A title that JSON must unescape. + await writeInfo(paths, "esc", { id: "esc", title: 'He said "hi" \\ bye ✓' }); + // A title beyond the 16 KB head: the full-parse fallback finds it. + await writeInfo(paths, "late", { id: "late", pad: "x".repeat(40_000), title: "Late title" }); + // The scan never creates a video dir, so reading must not either. + const got = await readChannelVideoTitles(paths, SLUG, ["esc", "late", "ghost"]); + assert.equal(got.get("esc")?.title, 'He said "hi" \\ bye ✓'); + assert.equal(got.get("late")?.title, "Late title"); + assert.equal(got.has("ghost"), false); + const dirs = await readdir(path.join(paths.channelsDir, SLUG, "data")); + assert.deepEqual(dirs.sort(), ["esc", "late"]); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("readVideoMetadataForDisplay: metadata.info.json, else the scan entry, else none", async () => { + const { dir, paths } = await fixture("vtitles-display-"); + try { + await writeInfo(paths, "dl", { + id: "dl", + title: "Downloaded", + description: "Full description", + webpage_url: "https://www.youtube.com/watch?v=dl", + uploader: "Uploader", + upload_date: "20250102", + duration: 125, + }); + await scan(paths, { + dl: { title: "Scan must lose" }, + un: { title: "Listed only", description: "Scan description", duration: 61 }, + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "dl"), { + title: "Downloaded", + description: "Full description", + webpageUrl: "https://www.youtube.com/watch?v=dl", + uploader: "Uploader", + uploadDate: "20250102", + duration: 125, + source: "metadata", + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "un"), { + title: "Listed only", + description: "Scan description", + uploadDate: "20260101", + duration: 61, + source: "scan", + }); + assert.deepEqual(await readVideoMetadataForDisplay(paths, SLUG, "nope"), { + source: "none", + }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +// The cost bar: a 5,000-id channel, all three sources exercised. Recorded in +// plans/release-8.md; asserted only loosely so a slow CI box does not flake. +test("cost: 5,000 ids across the three sources", async () => { + const { dir, paths } = await fixture("vtitles-cost-"); + try { + const ids = Array.from({ length: 5000 }, (_, i) => `v${String(i).padStart(5, "0")}`); + // 3,000 in the index, 1,500 in the scan, 400 on disk only, 100 bare. + await writeIndex( + paths.lmdbPath, + ids.slice(0, 3000).map((id, i) => ({ + slug: SLUG, + summary: summary(id, `Indexed ${id}`, `2025${String((i % 12) + 1).padStart(2, "0")}01`), + })), + ); + await scan( + paths, + Object.fromEntries( + ids.slice(3000, 4500).map((id) => [id, { title: `Scanned ${id}`, description: "x".repeat(500) }]), + ), + ); + for (const id of ids.slice(4500, 4900)) { + await writeInfo(paths, id, { id, title: `Disk ${id}`, formats: "f".repeat(50_000) }); + } + const t0 = performance.now(); + const got = await readChannelVideoTitles(paths, SLUG, ids); + const ms = performance.now() - t0; + assert.equal(got.size, 4900); + const bySource = { index: 0, scan: 0, metadata: 0 }; + for (const v of got.values()) bySource[v.source]++; + assert.deepEqual(bySource, { index: 3000, scan: 1500, metadata: 400 }); + const t1 = performance.now(); + await readChannelVideoTitles(paths, SLUG, ids); + const memoMs = performance.now() - t1; + console.log( + `readChannelVideoTitles 5,000 ids: ${ms.toFixed(1)} ms first, ${memoMs.toFixed(1)} ms memoized`, + ); + assert.ok(ms < 5000, `took ${ms} ms`); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("metadata titles are memoized per channel until data/ changes", async () => { + const { dir, paths } = await fixture("vtitles-memo-"); + try { + resetVideoTitleMemo(); + const dataDir = path.join(paths.channelsDir, SLUG, "data"); + await writeInfo(paths, "a", { id: "a", title: "Alpha" }); + await writeInfo(paths, "b", { id: "b", title: "Bravo" }); + // Pin the dir's mtime so the second call's key is certainly unchanged. + const pinned = new Date("2026-01-01T00:00:00Z"); + await utimes(dataDir, pinned, pinned); + + const r0 = videoTitleMetadataReadCount(); + const first = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.equal(first.get("a")?.title, "Alpha"); + assert.equal(videoTitleMetadataReadCount() - r0, 2); + + // Same data/ mtime: served from the memo, nothing read — even with the + // file gone (proof that it was not read). + await unlink(path.join(dataDir, "a", "metadata.info.json")); + await utimes(dataDir, pinned, pinned); + const r1 = videoTitleMetadataReadCount(); + const second = await readChannelVideoTitles(paths, SLUG, ["a", "b"]); + assert.equal(second.get("a")?.title, "Alpha"); + assert.equal(second.get("b")?.title, "Bravo"); + assert.equal(videoTitleMetadataReadCount() - r1, 0); + + // A new video dir changes data/'s mtime: the channel's memo is dropped and + // every id is read again ("a" now has no file, so no title). + await writeInfo(paths, "c", { id: "c", title: "Charlie" }); + const r2 = videoTitleMetadataReadCount(); + const third = await readChannelVideoTitles(paths, SLUG, ["a", "b", "c"]); + assert.equal(third.has("a"), false); + assert.equal(third.get("c")?.title, "Charlie"); + assert.equal(videoTitleMetadataReadCount() - r2, 3); + + // A miss is not memoized: a title that lands later is found. + await writeFile( + path.join(dataDir, "a", "metadata.info.json"), + JSON.stringify({ id: "a", title: "Alpha again" }), + ); + const fourth = await readChannelVideoTitles(paths, SLUG, ["a"]); + assert.equal(fourth.get("a")?.title, "Alpha again"); + } finally { + resetVideoTitleMemo(); + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/common/controller/videoTitles.ts b/common/controller/videoTitles.ts @@ -0,0 +1,318 @@ +// What a channel's videos are CALLED, for the editor's per-channel video list +// and the video page — without downloading anything and without walking the +// corpus. +// +// Three sources, cheapest first, first hit wins: +// +// 1. "index" — the LMDB `sums` sub-DB (TranscriptSummary.title), read by a +// key range over `byChannel` for this one channel. Covers the +// videos the index admitted, i.e. those with a +// metadata.info.json on disk at the last build. +// 2. "scan" — channels/<slug>/metadata-scan.json (metadataScanStore.ts): +// listed-but-undownloaded videos, only after a metadata scan. +// ONE file read for the whole channel. +// 3. "metadata" — data/<id>/metadata.info.json, for the remainder only +// (downloaded after the last index build, or a directory +// name that is not the index's metadata id — see below). A +// head read per id, not a full parse: some of these files +// are hundreds of KB. +// +// THE INDEX IS KEYED BY METADATA ID, THE LIST BY DIRECTORY NAME. On most +// channels they are the same string; on Rumble the directory is the URL slug +// and the metadata id is the embed id (recencyIndex.ts's layer-1 comment). A +// miss there is not wrong, it just falls through to source 3 and costs a file +// read. Nothing is matched fuzzily, so a title is never attributed to the +// wrong video. +// +// Read-only throughout: the index is opened `readOnly`, the scan store through +// its own loader, and no video directory is created (the scan store's +// invariant). + +import { existsSync } from "node:fs"; +import { open as openFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { open } from "lmdb"; +import type { Paths } from "../lib/paths"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { loadRawMetadataFromDir } from "../lib/transcripts-server"; +import { mapConcurrent } from "../lib/concurrency"; +import { loadMetadataScan } from "./metadataScanStore"; + +export type VideoTitleSource = "index" | "scan" | "metadata"; + +export type VideoTitle = { title: string; source: VideoTitleSource }; + +// buildIndex.ts's key shapes: sums is [uploadDate, slug, id], byChannel is +// [slug, uploadDate, id]. +type IndexKey = [string, string, string]; +type ChannelKey = [string, string, string]; + +// yt-dlp writes `"id"` then `"title"` first in metadata.info.json, so 16 KB of +// head finds the top-level title without reading the formats/subtitles tail. +const HEAD_BYTES = 16384; +const TITLE_RE = /"title":\s*("(?:[^"\\]|\\.)*")/; +const METADATA_READ_CONCURRENCY = 16; + +function channelDataDir(paths: Paths, slug: string): string { + return path.join(paths.channelsDir, slug, "data"); +} + +// Source 1. Never throws: a missing, locked or mid-rebuild index is "no +// titles from the index", and the other two sources carry on. +function readIndexTitles( + paths: Paths, + slug: string, + wanted: ReadonlySet<string>, + out: Map<string, VideoTitle>, +): void { + if (wanted.size === 0 || !existsSync(paths.lmdbPath)) return; + let root: ReturnType<typeof open>; + try { + // `compression: true` is not optional on a reader: buildIndex writes with + // it, and without it every value over ~1 KB (any real description) throws + // on decode. See curatedTagsPreview.ts's openIndex. + root = open({ + path: paths.lmdbPath, + readOnly: true, + maxDbs: 18, + compression: true, + }); + } catch { + return; + } + try { + const sums = root.openDB<TranscriptSummary, IndexKey>({ + name: "sums", + encoding: "msgpack", + }); + const byChannel = root.openDB<number, ChannelKey>({ + name: "byChannel", + encoding: "msgpack", + }); + // Key-only walk of this channel's range; a summary is decoded only for an + // id the caller asked about. + for (const { key } of byChannel.getRange({ + start: [slug], + end: [slug, "￿"], + })) { + const ck = key as ChannelKey; + if (ck[0] !== slug) break; + const id = ck[2]; + if (!wanted.has(id) || out.has(id)) continue; + const summary = sums.get([ck[1], ck[0], id]); + const title = summary?.title; + // summarize() falls back to the id when the metadata had no title — that + // is not a title, so leave the id for a later source. + if (typeof title === "string" && title.trim() && title !== id) { + out.set(id, { title, source: "index" }); + } + } + } catch { + // A partial read is still useful; whatever landed in `out` stands. + } finally { + void root.close().catch(() => {}); + } +} + +// Source 3, one id. A head read and a regex; a full parse only when the head +// did not contain a title (an unusual key order, or a pretty-printer that put +// it later). +async function readMetadataTitle(videoDir: string): Promise<string | null> { + const file = path.join(videoDir, "metadata.info.json"); + let head: string; + try { + const fh = await openFile(file, "r"); + try { + const buf = Buffer.alloc(HEAD_BYTES); + const { bytesRead } = await fh.read(buf, 0, HEAD_BYTES, 0); + head = buf.subarray(0, bytesRead).toString("utf8"); + } finally { + await fh.close(); + } + } catch { + return null; + } + const m = TITLE_RE.exec(head); + if (m) { + try { + const t = JSON.parse(m[1]) as unknown; + if (typeof t === "string" && t.trim()) return t; + } catch { + // fall through to the full parse + } + } + const meta = await loadRawMetadataFromDir(videoDir); + return typeof meta?.title === "string" && meta.title.trim() + ? meta.title + : null; +} + +// --- The source-3 memo -------------------------------------------------------- +// +// WHY: on a channel whose directory names never match the index (Rumble: dir = +// URL slug, index key = embed id; legacy `YYYYMMDD_<id>` dirs) EVERY row falls +// through to a metadata.info.json read, and the list page re-renders on every +// row click (force-dynamic, rows are `?video=` links). the-quartering-rumble has +// 8,049 such dirs: 5,024 ms cold / 224 ms warm per render, measured 2026-09-25. +// +// Per channel, keyed by the `data/` directory's mtime: adding or removing a +// video dir changes it and drops the channel's entry. Rewriting a +// metadata.info.json INSIDE an existing dir does not — accepted, because a +// video's title does not change after download. Misses (no file yet, e.g. a dir +// mid-download) are NOT memoized, so a title that lands later is picked up. +// Held on globalThis because Next can load this module more than once. +const MEMO_MAX_CHANNELS = 64; + +type TitleMemoEntry = { dataDirMtimeMs: number; titles: Map<string, string> }; + +declare global { + var __yttVideoTitleMemo__: Map<string, TitleMemoEntry> | undefined; +} + +function titleMemo(): Map<string, TitleMemoEntry> { + return (globalThis.__yttVideoTitleMemo__ ??= new Map()); +} + +// For tests and the e2e reset (api/test/invalidate-cache). +export function resetVideoTitleMemo(): void { + globalThis.__yttVideoTitleMemo__ = undefined; +} + +// Test hook: metadata.info.json reads so far in this process (tests diff it). +let metadataReads = 0; +export function videoTitleMetadataReadCount(): number { + return metadataReads; +} + +async function memoForChannel( + key: string, + dataDir: string, +): Promise<Map<string, string> | null> { + let mtimeMs: number; + try { + mtimeMs = (await stat(dataDir)).mtimeMs; + } catch { + return null; // no data/ — nothing to read, nothing to memoize + } + const memo = titleMemo(); + let entry = memo.get(key); + if (!entry || entry.dataDirMtimeMs !== mtimeMs) { + entry = { dataDirMtimeMs: mtimeMs, titles: new Map() }; + } + // Re-insert so iteration order is least-recently-used first. + memo.delete(key); + memo.set(key, entry); + while (memo.size > MEMO_MAX_CHANNELS) { + const oldest = memo.keys().next().value; + if (oldest === undefined) break; + memo.delete(oldest); + } + return entry.titles; +} + +// Titles for `ids` of one channel. An id with no title from any source is +// absent from the map — the caller shows the id. +export async function readChannelVideoTitles( + paths: Paths, + channelSlug: string, + ids: readonly string[], +): Promise<Map<string, VideoTitle>> { + const out = new Map<string, VideoTitle>(); + if (ids.length === 0) return out; + const wanted = new Set(ids); + + readIndexTitles(paths, channelSlug, wanted, out); + + if (out.size < wanted.size) { + const scan = await loadMetadataScan(paths, channelSlug); + for (const id of wanted) { + if (out.has(id)) continue; + const title = scan.entries[id]?.title; + if (title && title.trim()) out.set(id, { title, source: "scan" }); + } + } + + let remainder = [...wanted].filter((id) => !out.has(id)); + if (remainder.length > 0) { + const dataDir = channelDataDir(paths, channelSlug); + const memo = await memoForChannel(dataDir, dataDir); + if (!memo) return out; + remainder = remainder.filter((id) => { + const title = memo.get(id); + if (title === undefined) return true; + out.set(id, { title, source: "metadata" }); + return false; + }); + const titles = await mapConcurrent( + remainder, + METADATA_READ_CONCURRENCY, + (id) => { + metadataReads++; + return readMetadataTitle(path.join(dataDir, id)); + }, + ); + remainder.forEach((id, i) => { + const title = titles[i]; + if (title) { + memo.set(id, title); + out.set(id, { title, source: "metadata" }); + } + }); + } + return out; +} + +export type VideoDisplayMetadata = { + title?: string; + description?: string; + webpageUrl?: string; + uploader?: string; + uploadDate?: string; + duration?: number; + // "metadata" = data/<id>/metadata.info.json; "scan" = the channel's + // metadata-scan.json entry (a listed video that was never downloaded); + // "none" = neither exists, and the page shows the bare id. + source: "metadata" | "scan" | "none"; +}; + +function str(v: unknown): string | undefined { + return typeof v === "string" && v !== "" ? v : undefined; +} + +// The video page's header. metadata.info.json first — it is what the download +// wrote and carries the uploader and URL — else the scan entry. +export async function readVideoMetadataForDisplay( + paths: Paths, + channelSlug: string, + id: string, +): Promise<VideoDisplayMetadata> { + const meta = await loadRawMetadataFromDir( + path.join(channelDataDir(paths, channelSlug), id), + ); + if (meta) { + return { + title: str(meta.title), + description: str(meta.description), + webpageUrl: str(meta.webpage_url), + uploader: str(meta.uploader), + uploadDate: str(meta.upload_date), + duration: + typeof meta.duration === "number" && Number.isFinite(meta.duration) + ? meta.duration + : undefined, + source: "metadata", + }; + } + const scan = await loadMetadataScan(paths, channelSlug); + const entry = scan.entries[id]; + if (entry) { + return { + title: str(entry.title), + description: str(entry.description), + uploadDate: str(entry.uploadDate), + duration: entry.duration, + source: "scan", + }; + } + return { source: "none" }; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,8 @@ # Changelog ## [Unreleased] +- **A channel's video list shows titles, and you can search by them.** On `/channels/<slug>/videos`, each row now shows the video's title, with its id in smaller type underneath. Search matches the title or the id, ignoring case. The title comes from the transcript index for transcribed videos, from the channel's metadata scan for videos that were listed but never downloaded, and otherwise from the video's `metadata.info.json`. A video none of these name shows its id, as before. Each row's selection checkbox and link are named by the id as before, and the order is unchanged (by id). +- **An undownloaded video's page shows its title and details.** The video page used to show a bare id for any video without a `metadata.info.json`. If the channel's metadata scan has read the video, the page now shows its title, upload date and duration from the scan, marked *from the listing scan — not downloaded*. Any video page with a description, from either source, has a **Description** section, collapsed by default. - **New ops action: `pnpm ops keep-videos` marks every video of a channel whose title or description matches a pattern as "do not clean".** It sets the same marker as the video page's *Do not clean* toggle, so the clean sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge and saved-video eviction all leave those videos alone. The body is `{"slug", "match", "fields"?, "note"?, "dryRun"?}`. `match` is matched the way a channel's download filter *include* is: a case-insensitive regex over title + description. `fields: ["title"]` or `["description"]` narrows it to one half, and `dryRun: true` reports without writing. Videos that already carry the marker are counted and left as they are. The marker lives in the video's folder, so a match that was never downloaded is listed under `notDownloaded` and no folder is created for it. Run `download-missing` on those ids, then run `keep-videos` again. On a new channel, run `metadata-scan` first: a video with no scanned title cannot match, and the reply counts those as `unscanned`. - **Auto-download no longer retries the same rate-limited video over and over; it moves on to the next one.** When a download answered HTTP 429, the runner paused the whole platform for a while and then picked the same video again, because it was still first in the queue. Each retry doubled the pause, up to 30 minutes. On 2026-09-24 one YouTube Short was retried 12 times this way and kept YouTube paused all evening. A YouTube 429 comes from the subtitle fetch for one video, not from the whole site. Now a rate-limited video is also **deferred for 6 hours**: auto-download skips it, so when the pause ends the runner takes the next video. The pause still grows only when *different* videos keep hitting the limit. Deferrals are kept in `.auto-queue/state.json` beside the platform cooldowns, so a restart does not retry the video early. The log line reads `… (attempt 1). <id> deferred 6h; next video after cooldown.` A manual Sync or *download missing* ignores deferrals and still fetches the video. When every video left is deferred, the runner reports that it is idle for that reason: "every pending video was rate-limited recently and is deferred". - **The cooldown strip on `/operations/download` also lists deferred videos.** It is now a region named *Rate-limit cooldown*, with a *Platforms in cooldown* list (unchanged) and a *Deferred videos* list. Each deferred video links to its page and shows how long it has left (`alpha/a1 — 5h 59m left`). The strip appears when either list has something in it. Times over an hour now read `5h 59m` instead of `359m 58s`. diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts @@ -3,6 +3,7 @@ import { revalidatePath } from "next/cache"; import { resetSnapshotScheduler } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { resetChannelSnapshotMemo } from "yt-dlp-transcript-common/controller/channels"; import { resetStorageProbeMemo } from "yt-dlp-transcript-common/controller/storageLocations"; +import { resetVideoTitleMemo } from "yt-dlp-transcript-common/controller/videoTitles"; import { testRouteDenied } from "../_guard"; export const dynamic = "force-dynamic"; @@ -105,6 +106,11 @@ function invalidate() { // testInfo.outputPath varies by accident rather than on purpose; cleared here // so it is on purpose. resetStorageProbeMemo(); + // And the video list's metadata.info.json title memo. It is keyed by the + // channel's data/ mtime, which resetData() changes by recreating the dir — but + // a spec that rewrites a title in place inside one mtime tick would otherwise + // see the previous spec's title. + resetVideoTitleMemo(); revalidatePath("/", "layout"); return NextResponse.json({ ok: true }); } diff --git a/editor/app/channels/[slug]/components/VideoListPane.tsx b/editor/app/channels/[slug]/components/VideoListPane.tsx @@ -8,7 +8,11 @@ import { VirtualRow } from "yt-dlp-transcript-common/components/VirtualRow"; import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand"; import type { VideoRow, VideoFilter } from "../lib/videoRows"; import type { TagDef } from "../lib/videoTagRows"; -import { filterRows, serializeFilters } from "../lib/videoRows"; +import { + filterRows, + matchesVideoQuery, + serializeFilters, +} from "../lib/videoRows"; import { QueueControl } from "../../../components/QueueControl"; import { bulkClearFailedMarkersAction, @@ -147,9 +151,10 @@ export function VideoListPane({ (r.curatedTags ?? []).some((t) => tagFilter.has(t)), ); } - const q = query.trim().toLowerCase(); - if (!q) return filtered; - return filtered.filter((r) => r.id.toLowerCase().includes(q)); + if (!query.trim()) return filtered; + // Id OR title, case-insensitive — the same predicate the server applies to + // `?q=` for the detail pane's prev/next. + return filtered.filter((r) => matchesVideoQuery(r, query)); }, [rows, filters, query, tagFilter]); // VIRTUALIZED. The largest channel has ~11,000 videos, and one <li> + <Link> @@ -468,7 +473,7 @@ export function VideoListPane({ <div className="flex flex-col gap-2"> <input type="search" - placeholder="Search video id…" + placeholder="Search title or id…" value={query} onChange={(e) => changeQuery(e.target.value)} aria-label="search videos" @@ -651,7 +656,20 @@ export function VideoListPane({ aria-current={isActive ? "page" : undefined} > <StatusGlyphs row={row} /> - <span className="font-mono text-xs truncate">{row.id}</span> + {row.title ? ( + // Title first, id underneath: the id is still what the + // row's accessible name and the URL carry. + <span className="min-w-0 flex flex-col"> + <span className="text-sm truncate" title={row.title}> + {row.title} + </span> + <span className="font-mono text-[11px] text-muted-foreground truncate"> + {row.id} + </span> + </span> + ) : ( + <span className="font-mono text-xs truncate">{row.id}</span> + )} {row.running && ( <span aria-label="job running" diff --git a/editor/app/channels/[slug]/lib/videoRows.test.ts b/editor/app/channels/[slug]/lib/videoRows.test.ts @@ -0,0 +1,47 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { matchesVideoQuery, type VideoRow } from "./videoRows"; + +// The one predicate behind BOTH the list's search box (VideoListPane) and the +// server's `?q=` ordering for the detail pane's prev/next (videos/page.tsx). + +function row(id: string, title?: string): VideoRow { + return { + id, + ...(title ? { title } : {}), + downloaded: false, + transcribed: false, + untranscribable: false, + partial: false, + corruptSource: false, + corruptFullSource: false, + failedTranscription: false, + wrongFormatAudio: false, + excluded: false, + incompleteTranscript: false, + digestWarnings: false, + shortAudio: false, + running: false, + status: "not_downloaded", + }; +} + +test("matches the id, case-insensitively", () => { + assert.equal(matchesVideoQuery(row("AbC123xyz"), "abc123"), true); + assert.equal(matchesVideoQuery(row("AbC123xyz"), " XYZ "), true); + assert.equal(matchesVideoQuery(row("AbC123xyz"), "nope"), false); +}); + +test("matches the title, case-insensitively", () => { + const r = row("vid00000001", "Moonlit Harbor Interview"); + assert.equal(matchesVideoQuery(r, "harbor"), true); + assert.equal(matchesVideoQuery(r, "HARBOR interview"), true); + assert.equal(matchesVideoQuery(r, "vid0000"), true); + assert.equal(matchesVideoQuery(r, "sunrise"), false); +}); + +test("a row with no title matches on the id only; a blank query matches all", () => { + assert.equal(matchesVideoQuery(row("vid1"), "harbor"), false); + assert.equal(matchesVideoQuery(row("vid1"), ""), true); + assert.equal(matchesVideoQuery(row("vid1", "Anything"), " "), true); +}); diff --git a/editor/app/channels/[slug]/lib/videoRows.ts b/editor/app/channels/[slug]/lib/videoRows.ts @@ -11,6 +11,11 @@ export type VideoRowStatus = export type VideoRow = { id: string; + // What the video is called, when anything on disk says so — the transcript + // index, the channel's metadata scan, or data/<id>/metadata.info.json, in + // that order (common/controller/videoTitles.ts). Absent means the list shows + // the id alone. + title?: string; downloaded: boolean; transcribed: boolean; untranscribable: boolean; @@ -141,6 +146,16 @@ export function filterRows( ); } +// The search box: a case-insensitive substring of the id OR the title. +export function matchesVideoQuery(r: VideoRow, q: string): boolean { + const needle = q.trim().toLowerCase(); + if (!needle) return true; + return ( + r.id.toLowerCase().includes(needle) || + (r.title?.toLowerCase().includes(needle) ?? false) + ); +} + // Parse the comma-separated `?filter=` value into a set of valid filters, // ignoring unknown/legacy tokens (including "all"). A single legacy // `?filter=transcribed` parses to a one-element set, so old links still work. diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts @@ -26,6 +26,9 @@ export type ComputeRowsInput = { failedTranscriptionIds: string[]; runningJobs: JobRecord[]; excludedIds: Set<string>; + // id -> title, from readChannelVideoTitles. Optional so a caller that does + // not show titles need not pay for them. + titles?: ReadonlyMap<string, { title: string }>; }; export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { @@ -92,8 +95,10 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { else if (inDownloadedNoTranscript) status = "downloaded_no_transcript"; else status = "transcribed"; + const title = input.titles?.get(id)?.title; rows.push({ id, + ...(title ? { title } : {}), downloaded, transcribed, untranscribable: isUntranscribable, diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -2,7 +2,7 @@ import type { Metadata } from "next"; import Link from "next/link"; import { notFound } from "next/navigation"; import path from "node:path"; -import { readdir, readFile, stat } from "node:fs/promises"; +import { readdir, stat } from "node:fs/promises"; import type { Dirent } from "node:fs"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server"; @@ -13,6 +13,10 @@ import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/exclu import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { + readVideoMetadataForDisplay, + type VideoDisplayMetadata, +} from "yt-dlp-transcript-common/controller/videoTitles"; +import { isTranscriptVtt, resolvePrimaryVtt, } from "yt-dlp-transcript-common/lib/videoStatus"; @@ -38,14 +42,6 @@ import { loadVideoOperationPanels } from "./lib/videoOperationPanels"; export const dynamic = "force-dynamic"; -type VideoMeta = { - title?: string; - webpageUrl?: string; - uploader?: string; - uploadDate?: string; - duration?: number; -}; - async function loadVideoDir( slug: string, videoId: string, @@ -68,29 +64,11 @@ async function loadVideoDir( return { files }; } -async function loadMeta(slug: string, videoId: string): Promise<VideoMeta> { - const file = path.join( - getPaths().channelsDir, - slug, - "data", - videoId, - "metadata.info.json", - ); - try { - const raw = await readFile(file, "utf8"); - const j = JSON.parse(raw); - return { - title: typeof j?.title === "string" ? j.title : undefined, - webpageUrl: - typeof j?.webpage_url === "string" ? j.webpage_url : undefined, - uploader: typeof j?.uploader === "string" ? j.uploader : undefined, - uploadDate: - typeof j?.upload_date === "string" ? j.upload_date : undefined, - duration: typeof j?.duration === "number" ? j.duration : undefined, - }; - } catch { - return {}; - } +// metadata.info.json when the video was downloaded, else the channel's +// metadata-scan.json entry for a listed-but-never-downloaded video — so an +// undownloaded video's page is not a bare id. +function loadMeta(slug: string, videoId: string): Promise<VideoDisplayMetadata> { + return readVideoMetadataForDisplay(getPaths(), slug, videoId); } export async function generateMetadata({ @@ -216,7 +194,22 @@ export default async function VideoDetailPage({ {typeof meta.duration === "number" && ( <span>{formatDuration(meta.duration)}</span> )} + {meta.source === "scan" && ( + <span aria-label="metadata source" className="italic"> + from the listing scan — not downloaded + </span> + )} </div> + {meta.description && ( + <details className="text-sm"> + <summary className="cursor-pointer text-muted-foreground hover:text-foreground"> + Description + </summary> + <p className="mt-2 whitespace-pre-wrap break-words text-muted-foreground max-h-96 overflow-y-auto"> + {meta.description} + </p> + </details> + )} </header> <RunningJobsList jobs={activeJobs} hideChannelSlug hideVideoId /> diff --git a/editor/app/channels/[slug]/videos/page.tsx b/editor/app/channels/[slug]/videos/page.tsx @@ -1,5 +1,5 @@ import path from "node:path"; -import { readdir, readFile, stat } from "node:fs/promises"; +import { readdir, stat } from "node:fs/promises"; import type { Dirent } from "node:fs"; import type { Metadata } from "next"; import { notFound } from "next/navigation"; @@ -26,10 +26,16 @@ import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { VideoPanel, type VideoFile } from "../videos/[id]/components/VideoPanel"; import { VideoWorkspace } from "./components/VideoWorkspace"; import { readChannelConfigCached } from "../lib/channelConfigCache"; -import { parseFilters, filterRows, serializeFilters } from "../lib/videoRows"; +import { + parseFilters, + filterRows, + matchesVideoQuery, + serializeFilters, +} from "../lib/videoRows"; import { computeVideoRows, readDataDirVideoIds } from "../lib/videoRowsServer"; import { normalizeBuckets } from "yt-dlp-transcript-common/views/pipeline/stageStatus"; import { attachCuratedTags } from "../lib/videoTagRows"; +import { readChannelVideoTitles } from "yt-dlp-transcript-common/controller/videoTitles"; export const dynamic = "force-dynamic"; @@ -55,20 +61,6 @@ async function loadVideoDir( return { files }; } -async function loadVideoTitle( - channelDataDir: string, - videoId: string, -): Promise<string | null> { - const file = path.join(channelDataDir, videoId, "metadata.info.json"); - try { - const raw = await readFile(file, "utf8"); - const j = JSON.parse(raw); - return typeof j?.title === "string" ? j.title : null; - } catch { - return null; - } -} - export async function generateMetadata({ params, }: { @@ -140,12 +132,20 @@ export default async function ChannelVideosPage({ const channelDataDir = path.join(paths.channelsDir, slug, "data"); const channelDataDirIds = await readDataDirVideoIds(channelDataDir); + // What each video is called: one key range over the transcript index, one + // read of the channel's metadata-scan.json, and a head read of + // metadata.info.json only for what those two did not name. Measured at + // ~80 ms for a synthetic 5,000-id channel (plans/release-8.md, slice V). + const titles = await readChannelVideoTitles(paths, slug, [ + ...new Set([...channelDataDirIds, ...(snapshot.undownloadedIds ?? [])]), + ]); const rows = computeVideoRows({ channelDataDirIds, snapshot, failedTranscriptionIds: failedVideoIds, runningJobs, excludedIds: excludedDownloadIds, + titles, }); // Curated-tag pins, folded onto the rows so the list can offer Tag chips and @@ -166,9 +166,8 @@ export default async function ChannelVideosPage({ tagFilter.length === 0 || (r.curatedTags ?? []).some((t) => tagFilter.includes(t)), ); - const q = queryRaw.trim().toLowerCase(); - const orderedRows = q - ? filteredRows.filter((r) => r.id.toLowerCase().includes(q)) + const orderedRows = queryRaw.trim() + ? filteredRows.filter((r) => matchesVideoQuery(r, queryRaw)) : filteredRows; const selectedIndex = selectedVideoId ? orderedRows.findIndex((r) => r.id === selectedVideoId) @@ -192,7 +191,8 @@ export default async function ChannelVideosPage({ const [dirData, title, outcome, availabilityRecord, cov, excludedTrunc] = await Promise.all([ loadVideoDir(channelDataDir, selectedVideoId), - loadVideoTitle(channelDataDir, selectedVideoId), + // Already read for the list — the title map covers every row. + Promise.resolve(titles.get(selectedVideoId)?.title ?? null), loadDownloadOutcome(videoDir), loadAvailability(videoDir), readTranscriptCoverage(videoDir), diff --git a/editor/e2e/video-titles.spec.ts b/editor/e2e/video-titles.spec.ts @@ -0,0 +1,115 @@ +// The per-channel video list shows what each video is CALLED, and searches by +// it; an undownloaded video's own page is not a bare id. +// +// Three rows, one per title source that can reach a fixture (the transcript +// index is exercised by common/controller/videoTitles.test.ts): +// - fake00000001 — downloaded; its title is in data/<id>/metadata.info.json. +// - fake00000002 — listed, never downloaded; titled by the channel's +// metadata-scan.json (the listing scan's store). +// - fake00000003 — listed, never downloaded, never scanned: the bare id. + +import { writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { + channelVideos, + generateReport, + resetData, + resolvePath, +} from "./helpers"; + +const CHANNEL = "test-youtube"; +const ROOT = `test-transcripts/channels/${CHANNEL}`; +const DOWNLOADED = "fake00000001"; +const SCANNED = "fake00000002"; +const BARE = "fake00000003"; +const SCANNED_TITLE = "Moonlit Harbor Interview"; +const SCANNED_DESCRIPTION = "A long talk on the pier.\nSecond line of notes."; + +async function seed() { + await resetData("youtube-with-playlist"); + // The store's on-disk shape (common/controller/metadataScanStore.ts, + // METADATA_SCAN_FILENAME) — the scan writes exactly this, and never a + // data/<id>/ directory. + await writeFile( + resolvePath(`${ROOT}/metadata-scan.json`), + JSON.stringify( + { + version: 1, + updatedAt: "2026-09-25T00:00:00.000Z", + lastRun: null, + entries: { + [SCANNED]: { + title: SCANNED_TITLE, + description: SCANNED_DESCRIPTION, + uploadDate: "20240315", + duration: 754, + scannedAt: "2026-09-25T00:00:00.000Z", + }, + }, + errors: {}, + }, + null, + 2, + ), + ); +} + +test("the video list shows titles and searches them", async ({ page }) => { + await seed(); + await generateReport(page, CHANNEL); + await page.goto(channelVideos(CHANNEL)); + + const list = page.getByLabel("videos", { exact: true }); + const downloaded = list.getByLabel(`open ${DOWNLOADED}`); + const scanned = list.getByLabel(`open ${SCANNED}`); + const bare = list.getByLabel(`open ${BARE}`); + // Titles where known, the id beneath them; the bare id alone otherwise. + await expect(downloaded).toContainText("Synthetic Test Video 1"); + await expect(downloaded).toContainText(DOWNLOADED); + await expect(scanned).toContainText(SCANNED_TITLE); + await expect(scanned).toContainText(SCANNED); + await expect(bare).toHaveText(BARE); + + // A word from the scan-store title, in the wrong case, leaves that one row. + const search = page.getByLabel("search videos"); + await search.fill("harbor"); + await expect(scanned).toBeVisible(); + await expect(list.getByLabel(/^open fake/)).toHaveCount(1); + + // The id still matches, as it always has. + await search.fill(BARE); + await expect(bare).toBeVisible(); + await expect(list.getByLabel(/^open fake/)).toHaveCount(1); +}); + +test("an undownloaded video's page shows the listing scan's metadata", async ({ + page, +}) => { + await seed(); + await page.goto(`/channels/${CHANNEL}/videos/${SCANNED}`); + + await expect( + page.getByRole("heading", { name: SCANNED_TITLE }), + ).toBeVisible(); + await expect(page.getByLabel("metadata source")).toHaveText( + "from the listing scan — not downloaded", + ); + // Date and duration come from the scan entry too. + await expect(page.getByText("2024-03-15")).toBeVisible(); + await expect(page.getByText("12:34")).toBeVisible(); + + // Collapsed by default; opening it shows the description. + const summary = page.getByText("Description", { exact: true }); + await expect(summary).toBeVisible(); + const body = page.getByText("A long talk on the pier."); + await expect(body).toBeHidden(); + await summary.click(); + await expect(body).toBeVisible(); + + // A downloaded video's page has no scan note. + await page.goto(`/channels/${CHANNEL}/videos/${DOWNLOADED}`); + await expect( + page.getByRole("heading", { name: "Synthetic Test Video 1" }), + ).toBeVisible(); + await expect(page.getByLabel("metadata source")).toHaveCount(0); +}); diff --git a/plans/release-8.md b/plans/release-8.md @@ -0,0 +1,108 @@ +# Release 8 — video titles (+ follow-ups) + +`main` at `bb3dbb4c`, release 7 live 2026-09-25; this release starts with the operator's video-titles +ask. Rules: `plans/tools/implementer-rules.md`. Record file: this file. + +## Record + +### Slice V, as shipped — video titles (2026-09-25) + +Branch `one-core/r8-video-titles` off `main` `bb3dbb4c`. The operator's ask (2026-09-25): "In the +editor, the individual video view should show video metadata like the title, and `/videos` should +show titles in the list when available and allow searching by them." There is no global `/videos`; +the list is the per-channel workspace `editor/app/channels/[slug]/videos/page.tsx`. Before this +slice a list row was built from `ChannelSnapshot` id lists and showed only the id, the search box +matched only the id, and the video page read `metadata.info.json` with its own parser, so a video +that was never downloaded showed a bare id even when the metadata scan knew what it was called. + +Three places hold a title, and the slice reads them cheapest first. The first one found wins: +- **`index`**: the LMDB `sums` sub-DB, reached by a key-only walk of this channel's `byChannel` + range. A summary is decoded only for an id the list asked about. It is opened `readOnly` **with + `compression: true`**, for the reason `curatedTagsPreview.ts`'s `openIndex` gives: without it, + any value over ~1 KB throws. A missing or unopenable index gives no titles and is not an error. + An index title equal to the id is `summarize()`'s fallback, not a title, and is skipped. +- **`scan`**: `channels/<slug>/metadata-scan.json`, read once through `loadMetadataScan`. +- **`metadata`**: `data/<id>/metadata.info.json`, only for ids the first two did not name. It is a + 16 KB head read plus a regex (yt-dlp writes `id` then `title` first). The full parse + (`loadRawMetadataFromDir`) runs only when the head has no title. 16 reads run concurrently. + +Nothing creates `data/<id>/`, which keeps the scan store's invariant. `metadataScanStore.ts` and +`channelSnapshot.ts` are unchanged. + +| sha | what | +|---|---| +| `9ed6cce8` | `common/controller/videoTitles.ts`: `readChannelVideoTitles(paths, slug, ids) → Map<id, {title, source: "index"\|"scan"\|"metadata"}>` and `readVideoMetadataForDisplay(paths, slug, id) → {title?, description?, webpageUrl?, uploader?, uploadDate?, duration?, source: "metadata"\|"scan"\|"none"}`. `.test.ts` has 6 cases on a real compressed LMDB with a 4 KB description: the three-source merge with first hit winning, bare ids absent and another channel's id not leaking; missing index and scan store; the id-as-title fallback skipped; the head read with an escaped title, a title past 16 KB and no dir created; the display reader's three outcomes; and the 5,000-id cost case | +| `cdd6c3dc` | The list. `VideoRow.title?`. `computeVideoRows` takes an optional `titles` map, which the page fills once from `readChannelVideoTitles` over data-dir ids ∪ `undownloadedIds`. `matchesVideoQuery` (id OR title, case-insensitive) is used by both the client filter and the server's `?q=` ordering for prev/next. A row shows the title with the id on a muted mono line under it, and shows the id alone when there is no title. **The row's accessible name stays `open <id>`**, and the checkbox stays `select <id>`. The placeholder is now "Search title or id…". The empty copy ("No videos match this filter.") names no ids and is unchanged. The embedded detail pane's `video title` line reads the same map, and the page's private `loadVideoTitle` parser is gone | +| `16ef0758` | The video page. `loadMeta` is now `readVideoMetadataForDisplay`, and the page's private parser is gone. When the source is `scan`, the header shows the scan's title, upload date and duration, plus an italic note "from the listing scan — not downloaded" (`aria-label="metadata source"`). A collapsed `<details>` **Description** block appears when either source has a description. `generateMetadata` behaves as before (title, else id) and now also finds scan titles. New `editor/e2e/video-titles.spec.ts` (2 tests) on `youtube-with-playlist` plus a seeded `metadata-scan.json`: `fake00000001` shows its `metadata.info.json` title, `fake00000002` its scan title, and `fake00000003` its bare id. "harbor" leaves one row, and the id still matches. The undownloaded page shows the h2, the scan note, `2024-03-15`, `12:34`, and a Description that is collapsed and opens on click. The downloaded page has no scan note | +| `ff3944cd` | this record and two `[Unreleased]` bullets in `editor/CHANGELOG.md` | +| `4e8bb285` | **Review M1.** Source 3 is memoized per channel in `videoTitles.ts` (`globalThis.__yttVideoTitleMemo__`: `Map<dataDir, {dataDirMtimeMs, titles}>`). Each render does one `stat` of `data/`. A changed mtime (a video dir added or removed) drops the channel's entry. Only ids missing from the memo are read, and misses are not memoized, so a dir still being downloaded picks up its title once the file lands. The memo holds at most 64 channels, least recently used evicted first. `resetVideoTitleMemo()` is called by `api/test/invalidate-cache` alongside the other process memos. `videoTitleMetadataReadCount()` is a test hook. New unit case: with `data/`'s mtime unchanged, the second call reads 0 files and still serves a title whose file was deleted; a new dir makes it read all 3 again; a miss is re-read later | +| `41facfe0` | **Review L1 + L2.** `matchesVideoQuery` moves above `parseFilters`' doc comment. `editor/app/channels/[slug]/lib/videoRows.test.ts` (3 cases) covers id hits, title hits (case-insensitive), a row without a title, and a blank query. This pins the server `?q=` path, which shares the predicate with the search box | +| *(this commit)* | the re-gate below and the corrected cost paragraph | + +**Gates** (worktree root, on `16ef0758`). tsc (`pnpm -r --no-bail --workspace-concurrency=1 exec +tsc --noEmit`) was clean before each commit. common **1801/1801** = 1795 + 6 (`videoTitles`). +Editor unit **72/72**. test:scripts **161 pass + 1 skip**. mcp **219/219**. `pnpm --filter editor +exec next build` ok. The export build was not run, because the slice touches no file under +`export/` and no `common/` module it imports. EDITOR e2e used `$T/v-specs.txt`. `videos.spec` and +`channel-page.spec` do not exist, so the list is `video-titles`, `video-page`, +`video-filter-combine`, and the six specs that drive the list or the embedded pane +(`channel-embedded-video`, `bulk-actions`, `incomplete-transcript`, `download-format-guard`, +`channel-storage`, `whisper`). Result (`v-e2e.log`): **62 passed, 0 failed, 4.0 m**, after a +2 m 38 s queue wait behind `one-core/r8-state-share`. The run used this worktree's block +(`PORT=4011`, `EXPORT_PORT=4010`); the `PORT:3011` printed by the e2e script is a default for when +the env var is unset. Every heavy step started with ≥ 3 GB available. + +**Re-gate after the review fixes** (on `41facfe0`). tsc was clean. common **1802/1802** = 1801 + 1 +(memo case). Editor unit **75/75** = 72 + 3 (`videoRows.test.ts`). `pnpm --filter editor exec next +build` ok. The memo touches the e2e reset route, so the EDITOR e2e re-ran the full nine-spec list +(`v-e2e2.log`): **62 passed, 0 failed, 3.6 m**, with no queue wait. test:scripts and mcp were not +re-run, because neither imports anything that changed. + +**Numbers: none.** No `settings.json`, `site.json` or `config.json` key changed, and nothing is +written: the slice only reads. + +**Cost of the title map** (the brief's bar: "does not change the page's order of magnitude"). +The first version held that bar only for a YouTube-shaped channel, where most titles come from the +index. The review found the Rumble case. On Rumble every directory name misses the index, so every +row fell through to a `metadata.info.json` head read, on **every** render. Each row click +re-renders, because rows are `?video=` links on a force-dynamic page. `the-quartering-rumble` has +**8,049** dirs. The reviewer replayed the read pattern read-only and measured **5,024 ms cold / +224 ms warm per render** before the memo. After the memo (`4e8bb285`), measured read-only with +`readChannelVideoTitles` itself over the live `data/` (no index opened, `$T/v-memo-measure.mts`): +the **first render took 5,331 ms** (cold page cache, all 8,049 titled) and **memoized renders took +5.5 ms and 6.5 ms**. The cold first render is still paid once per editor process, and again after a +video dir is added or removed on that channel. In the unit test's synthetic 5,000-id mix (3,000 +index, 1,500 scan, 400 `metadata.info.json`, 100 bare) it is 82–148 ms first and ~75 ms memoized. +That run is dominated by the index walk and the scan store, which are not memoized: one LMDB range +and one JSON read, not N file reads. + +**The memo's one blind spot, accepted.** A `metadata.info.json` rewritten inside an *existing* video +dir does not change `data/`'s mtime, so the list keeps the old title until the editor restarts or a +video dir is added or removed. A video's title does not change after download, so this was +accepted. + +**Found and left.** +- **Coverage is partial on `paramount-tactical`.** A read-only `jq` over live `snapshot.json` / + `metadata-scan.json` found one channel with undownloaded ids: `paramount-tactical`, with **1,439** + and no `metadata-scan.json`. Those rows show ids until a metadata scan runs, which is the same + step slice K's `keep-videos` needs. Every other channel's list is fully downloaded, so it is + titled from the index and `metadata.info.json`. +- **The index is keyed by metadata id and the list by directory name.** On Rumble (URL-slug dirs, + embed-id metadata) and on legacy `YYYYMMDD_<id>` dirs, the index lookup misses. Those rows are + titled from `metadata.info.json`, and the memo above now carries that cost. Matching dir names to + index ids (for example through the `mtimes` sub-DB's `[slug, videoDir]` keys) would make the cold + first render cheap too, but is left for later. +- **L3 (review, left).** When a previously scanned video has a `data/<id>/` dir but no readable + `metadata.info.json` (a failed or partial download), the page falls back to the scan entry and says + "not downloaded" beside a files panel. This is rare, and the wording was left as it is. +- **Rows stay sorted by id**, as before. Sorting by title or date is a separate ask. +- `VideoListPane.tsx` lives at `editor/app/channels/[slug]/components/`, not under `videos/**`. The + brief names "the list pane" in ownership, so it was edited as in scope. `ROW_ESTIMATE_PX` (30) was + left alone. Titled rows are about 40 px, and the virtualizer measures each row, so only the first + scrollbar estimate is off. +- The stage lists (`VideoIdList.tsx`, on the download/transcribe stage panels) still show bare ids. + They are not the `/videos` list, and titling them is outside this slice. +- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as the release-6 and release-7 + implementers did. + +## Rollout