commit d5cb49794a3ebc589dbcc2bc70409bc0d0e9a7f3
parent 552a70c17bb05455caa3e21be7dd7e89eff79343
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 21 Apr 2026 19:16:14 -0400
Use pages for transcripts
Diffstat:
5 files changed, 344 insertions(+), 21 deletions(-)
diff --git a/app/transcriptCache.ts b/app/transcriptCache.ts
@@ -1,10 +1,17 @@
"use client";
import type { TranscriptDetail } from "@/lib/transcripts";
+import type { ChannelTranscriptsManifest } from "@/lib/manifest";
+import { transcriptPageFileName } from "@/lib/manifest";
import { idbGet, idbPutBatch } from "./transcriptStore";
const resolved = new Map<string, TranscriptDetail>();
const inFlight = new Map<string, Promise<TranscriptDetail>>();
+const channelManifests = new Map<
+ string,
+ Promise<ChannelTranscriptsManifest>
+>();
+const pagePromises = new Map<string, Promise<TranscriptDetail[]>>();
export function fetchTranscript(slug: string): Promise<TranscriptDetail> {
const hit = resolved.get(slug);
@@ -24,12 +31,69 @@ export function fetchTranscript(slug: string): Promise<TranscriptDetail> {
return p;
}
+function fetchChannelManifest(
+ channelSlug: string,
+): Promise<ChannelTranscriptsManifest> {
+ let p = channelManifests.get(channelSlug);
+ if (!p) {
+ p = fetch(`/transcripts/${channelSlug}/manifest.json`).then((r) => {
+ if (!r.ok)
+ throw new Error(
+ `Failed to fetch transcripts manifest for ${channelSlug}`,
+ );
+ return r.json() as Promise<ChannelTranscriptsManifest>;
+ });
+ p.catch(() => channelManifests.delete(channelSlug));
+ channelManifests.set(channelSlug, p);
+ }
+ return p;
+}
+
+function fetchPage(
+ channelSlug: string,
+ pageIndex: number,
+): Promise<TranscriptDetail[]> {
+ const key = `${channelSlug}:${pageIndex}`;
+ let p = pagePromises.get(key);
+ if (!p) {
+ p = fetch(
+ `/transcripts/${channelSlug}/${transcriptPageFileName(pageIndex)}`,
+ ).then((r) => {
+ if (!r.ok)
+ throw new Error(
+ `Failed to fetch transcript page ${channelSlug}/${pageIndex}`,
+ );
+ return r.json() as Promise<TranscriptDetail[]>;
+ });
+ p.catch(() => pagePromises.delete(key));
+ pagePromises.set(key, p);
+ }
+ return p;
+}
+
async function load(slug: string): Promise<TranscriptDetail> {
const stored = await idbGet(slug);
if (stored) return stored;
- const r = await fetch(`/transcripts/${slug}.json`);
- if (!r.ok) throw new Error(`Failed to fetch transcript ${slug}`);
- const detail = (await r.json()) as TranscriptDetail;
- idbPutBatch(detail);
- return detail;
+ const slashIdx = slug.indexOf("/");
+ if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`);
+ const channelSlug = slug.slice(0, slashIdx);
+ const videoId = slug.slice(slashIdx + 1);
+ const manifest = await fetchChannelManifest(channelSlug);
+ const pageIndex = manifest.slugToPage[videoId];
+ if (pageIndex === undefined)
+ throw new Error(`Unknown transcript slug: ${slug}`);
+ const page = await fetchPage(channelSlug, pageIndex);
+ let found: TranscriptDetail | undefined;
+ for (const entry of page) {
+ if (entry.slug === slug) found = entry;
+ // Opportunistically warm the per-slug memory + IDB caches so subsequent
+ // fetches for other videos in this page hit without a network round-trip.
+ // Huge win for searchPipeline which iterates many slugs from the same
+ // channel/page.
+ resolved.set(entry.slug, entry);
+ idbPutBatch(entry);
+ }
+ if (!found)
+ throw new Error(`Transcript ${slug} missing from page ${pageIndex}`);
+ return found;
}
diff --git a/lib/manifest.ts b/lib/manifest.ts
@@ -18,3 +18,18 @@ export const SUMMARIES_PAGE_SIZE = 1000;
export function pageFileName(index: number): string {
return `page-${String(index).padStart(4, "0")}.json`;
}
+
+export const TRANSCRIPTS_MANIFEST_VERSION = 1;
+
+export function transcriptPageFileName(index: number): string {
+ return `page-${String(index).padStart(4, "0")}.json`;
+}
+
+export type ChannelTranscriptsManifest = {
+ version: number;
+ channelSlug: string;
+ pageCount: number;
+ maxPageBytes: number;
+ generatedAt: string;
+ slugToPage: Record<string, number>;
+};
diff --git a/lib/settings.ts b/lib/settings.ts
@@ -6,13 +6,19 @@ export type SiteSettings = {
siteDescription: string;
headerTitle: string;
homeTagline: string;
+ maxTranscriptPageBytes: number;
};
+export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
+export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
+export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
+
const DEFAULTS: SiteSettings = {
siteTitle: "Transcript Browser",
siteDescription: "Browse and search video transcripts",
headerTitle: "Transcript Browser",
homeTagline: "",
+ maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES,
};
let cached: SiteSettings | null = null;
@@ -26,6 +32,18 @@ export function getSettings(): SiteSettings {
} catch {
parsed = {};
}
- cached = { ...DEFAULTS, ...parsed };
+ const merged = { ...DEFAULTS, ...parsed };
+ merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes);
+ cached = merged;
return cached;
}
+
+function clampPageBytes(value: unknown): number {
+ const n =
+ typeof value === "number" && Number.isFinite(value)
+ ? value
+ : TRANSCRIPT_PAGE_DEFAULT_BYTES;
+ if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES;
+ if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES;
+ return Math.floor(n);
+}
diff --git a/lib/transcripts.ts b/lib/transcripts.ts
@@ -45,6 +45,8 @@ export type TranscriptDetail = TranscriptSummary & {
cues: Cue[] | undefined;
};
+export type TranscriptPage = TranscriptDetail[];
+
export async function countTranscripts(): Promise<number> {
const raw = await readFile(MANIFEST_PATH, "utf8");
const parsed = JSON.parse(raw) as { totalCount?: number };
diff --git a/scripts/build-index.ts b/scripts/build-index.ts
@@ -1,13 +1,17 @@
#!/usr/bin/env tsx
// Preprocess transcripts/channels/<channelSlug>/data/<videoDir>/ into:
// - LMDB cache at transcripts/index.mdb (for incremental rebuilds)
-// - public/summaries/{manifest,page-NNNN}.json (paginated summaries index)
-// - public/transcripts/<channelSlug>/<id>.json (per-transcript cues + metadata)
+// - public/summaries/{manifest,page-NNNN}.json (paginated cross-channel
+// summaries index for browse/search)
+// - public/transcripts/<channelSlug>/{manifest,page-NNNN}.json
+// (per-channel paginated transcript detail + cues, oldest-first so
+// adding a newer video only dirties the last page)
//
// Per-channel config.json selects the transcript parser ("youtube" → VTT,
// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match.
import path from "node:path";
+import { createHash } from "node:crypto";
import {
mkdir,
readdir,
@@ -26,13 +30,24 @@ import {
toDisplaySummary,
type RawMetadata,
} from "../lib/transcripts-server";
-import type { TranscriptSummary, DisplaySummary } from "../lib/transcripts";
-import type { Manifest, ChannelEntry } from "../lib/manifest";
+import type {
+ TranscriptSummary,
+ TranscriptDetail,
+ DisplaySummary,
+} from "../lib/transcripts";
+import type {
+ Manifest,
+ ChannelEntry,
+ ChannelTranscriptsManifest,
+} from "../lib/manifest";
import {
MANIFEST_VERSION,
SUMMARIES_PAGE_SIZE,
+ TRANSCRIPTS_MANIFEST_VERSION,
pageFileName,
+ transcriptPageFileName,
} from "../lib/manifest";
+import { getSettings } from "../lib/settings";
const ROOT = process.cwd();
const CHANNELS_DIR = path.join(ROOT, "transcripts", "channels");
@@ -42,7 +57,7 @@ const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries");
const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts");
const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json");
-const SCHEMA_VERSION = 3;
+const SCHEMA_VERSION = 4;
type Handling = "youtube" | "transcribe";
type ChannelConfig = { handling: Handling; name?: string };
@@ -50,6 +65,10 @@ type ChannelConfig = { handling: Handling; name?: string };
// Primary sort/lookup key: [uploadDate, channelSlug, id]. Reverse iteration
// yields newest-first directly (uploadDate is YYYYMMDD).
type IndexKey = [string, string, string];
+// Per-channel forward-scan index: [channelSlug, uploadDate, id].
+type ChannelKey = [string, string, string];
+// Page-hash key: [channelSlug, pageIndex].
+type PageHashKey = [string, number];
// Path key: [channelSlug, videoDir]. Stable regardless of metadata contents,
// so diffing by mtime doesn't require parsing metadata.info.json.
type PathKey = [string, string];
@@ -60,6 +79,16 @@ type MtimeRecord = {
indexKey: IndexKey;
};
+type PageHashRecord = {
+ hash: string;
+ entryCount: number;
+ sizeBytes: number;
+};
+
+function indexToChannelKey(k: IndexKey): ChannelKey {
+ return [k[1], k[0], k[2]];
+}
+
type LiveEntry = {
channelSlug: string;
handling: Handling;
@@ -192,6 +221,20 @@ async function main(): Promise<void> {
name: "mtimes",
encoding: "msgpack",
});
+ // Per-channel forward index: lets us stream [channelSlug, uploadDate, id]
+ // in oldest-first order for size-based page packing without loading every
+ // channel's entries into memory.
+ const byChannel = root.openDB<number, ChannelKey>({
+ name: "byChannel",
+ encoding: "msgpack",
+ });
+ // Content hashes of emitted transcript pages, so unchanged pages skip the
+ // write. Since page 1 = oldest videos, adding a new newest-upload video
+ // only dirties the last page's hash.
+ const pageHashes = root.openDB<PageHashRecord, PageHashKey>({
+ name: "pageHashes",
+ encoding: "msgpack",
+ });
const meta = root.openDB<unknown, string>({
name: "meta",
encoding: "msgpack",
@@ -206,6 +249,8 @@ async function main(): Promise<void> {
await sums.clearAsync();
await cues.clearAsync();
await mtimes.clearAsync();
+ await byChannel.clearAsync();
+ await pageHashes.clearAsync();
await meta.put("schema", SCHEMA_VERSION);
}
@@ -245,7 +290,9 @@ async function main(): Promise<void> {
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
- // Short-circuit: nothing changed AND public/ is intact.
+ // Short-circuit: nothing changed AND public/ is intact AND the configured
+ // max page size matches what each channel was last paginated with.
+ const { maxTranscriptPageBytes: configuredMaxPageBytes } = getSettings();
if (!anyMutations && !schemaBumped) {
const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null);
if (manifestRaw) {
@@ -261,7 +308,37 @@ async function main(): Promise<void> {
SUMMARIES_DIR,
pageFileName(Math.max(0, parsed.pageCount - 1)),
);
- if ((await exists(firstPage)) && (await exists(lastPage))) {
+ let transcriptsIntact = true;
+ for (const channelSlug of channelConfigs.keys()) {
+ const mPath = path.join(
+ TRANSCRIPTS_DIR,
+ channelSlug,
+ "manifest.json",
+ );
+ const raw = await readFile(mPath, "utf8").catch(() => null);
+ if (!raw) {
+ transcriptsIntact = false;
+ break;
+ }
+ try {
+ const cm = JSON.parse(raw) as ChannelTranscriptsManifest;
+ if (
+ cm.version !== TRANSCRIPTS_MANIFEST_VERSION ||
+ cm.maxPageBytes !== configuredMaxPageBytes
+ ) {
+ transcriptsIntact = false;
+ break;
+ }
+ } catch {
+ transcriptsIntact = false;
+ break;
+ }
+ }
+ if (
+ transcriptsIntact &&
+ (await exists(firstPage)) &&
+ (await exists(lastPage))
+ ) {
console.log(
`Index up to date (${live.length} transcripts, ${parsed.pageCount} pages). Skipping.`,
);
@@ -330,22 +407,18 @@ async function main(): Promise<void> {
if (prev && !indexKeysEqual(prev.indexKey, indexKey)) {
sums.remove(prev.indexKey);
cues.remove(prev.indexKey);
+ byChannel.remove(indexToChannelKey(prev.indexKey));
}
sums.put(indexKey, summary);
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
+ byChannel.put(indexToChannelKey(indexKey), 1);
mtimes.put(pk, {
metaMs: s.metaMs,
transcriptMs: s.transcriptMs,
indexKey,
});
-
- const detail = { ...summary, cues: cueList };
- await writeJsonAtomic(
- path.join(TRANSCRIPTS_DIR, `${summary.slug}.json`),
- detail,
- );
} catch (err) {
console.warn(
`Failed to process ${s.channelSlug}/${s.videoDir}:`,
@@ -365,15 +438,166 @@ async function main(): Promise<void> {
for (const { pathKey, indexKey } of removed) {
sums.remove(indexKey);
cues.remove(indexKey);
+ byChannel.remove(indexToChannelKey(indexKey));
mtimes.remove(pathKey);
- const slug = `${indexKey[1]}/${indexKey[2]}`;
- await rm(path.join(TRANSCRIPTS_DIR, `${slug}.json`), { force: true });
}
await sums.flushed;
await cues.flushed;
+ await byChannel.flushed;
await mtimes.flushed;
+ // Emit per-channel paginated transcript pages. Pages are size-packed
+ // oldest-first, so newer videos land on the last page and older pages
+ // hash-match from build to build.
+ const maxTranscriptPageBytes = configuredMaxPageBytes;
+ const generatedAt = new Date().toISOString();
+ let pagesWritten = 0;
+ let pagesSkipped = 0;
+ let pagesDeleted = 0;
+
+ type PendingEntry = { encoded: string; id: string };
+
+ const writePage = async (
+ channelSlug: string,
+ idx: number,
+ entries: PendingEntry[],
+ ): Promise<void> => {
+ const body = `[${entries.map((e) => e.encoded).join(",")}]`;
+ const hash = createHash("sha1").update(body).digest("hex");
+ const prev = pageHashes.get([channelSlug, idx]);
+ const outPath = path.join(
+ TRANSCRIPTS_DIR,
+ channelSlug,
+ transcriptPageFileName(idx),
+ );
+ if (prev?.hash === hash && (await exists(outPath))) {
+ pagesSkipped++;
+ return;
+ }
+ const sizeBytes = Buffer.byteLength(body, "utf8");
+ const tmp = `${outPath}.tmp-${process.pid}`;
+ await writeFile(tmp, body);
+ await rename(tmp, outPath);
+ pageHashes.put([channelSlug, idx], {
+ hash,
+ entryCount: entries.length,
+ sizeBytes,
+ });
+ pagesWritten++;
+ };
+
+ for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ const channelDir = path.join(TRANSCRIPTS_DIR, channelSlug);
+ await mkdir(channelDir, { recursive: true });
+
+ let pageIdx = 0;
+ let buffer: PendingEntry[] = [];
+ // Running payload size excluding the outer `[` and `]` (accounted at
+ // flush time). Includes the leading commas between elements.
+ let payloadBytes = 0;
+ const slugToPage: Record<string, number> = {};
+
+ for (const { key } of byChannel.getRange({
+ start: [channelSlug],
+ end: [channelSlug, ""],
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== channelSlug) continue;
+ const indexKey: IndexKey = [ck[1], ck[0], ck[2]];
+ const summary = sums.get(indexKey);
+ if (!summary) continue;
+ const cueList = cues.get(indexKey);
+ const detail: TranscriptDetail = { ...summary, cues: cueList };
+ const encoded = JSON.stringify(detail);
+ const entryBytes = Buffer.byteLength(encoded, "utf8");
+ const commaBytes = buffer.length === 0 ? 0 : 1;
+ const delta = entryBytes + commaBytes;
+
+ // Close current page if this entry would push it past the limit.
+ // `+ 2` accounts for the outer brackets.
+ if (buffer.length > 0 && payloadBytes + delta + 2 > maxTranscriptPageBytes) {
+ await writePage(channelSlug, pageIdx, buffer);
+ pageIdx++;
+ buffer = [];
+ payloadBytes = 0;
+ }
+
+ if (entryBytes + 2 > maxTranscriptPageBytes) {
+ console.warn(
+ ` ${summary.slug}: entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`,
+ );
+ }
+
+ slugToPage[summary.id] = pageIdx;
+ buffer.push({ encoded, id: summary.id });
+ payloadBytes += delta;
+ }
+
+ if (buffer.length > 0) {
+ await writePage(channelSlug, pageIdx, buffer);
+ pageIdx++;
+ }
+
+ const pageCount = pageIdx;
+
+ // Prune stale per-channel page files + drop pageHashes beyond pageCount.
+ const keep = new Set<string>(["manifest.json"]);
+ for (let i = 0; i < pageCount; i++) keep.add(transcriptPageFileName(i));
+ const existing = await readdir(channelDir).catch(() => [] as string[]);
+ for (const name of existing) {
+ if (keep.has(name)) continue;
+ await rm(path.join(channelDir, name), { force: true });
+ pagesDeleted++;
+ }
+ // pageHashes may still have records for page indexes >= pageCount
+ // (channel shrank). Drop them.
+ for (const { key } of pageHashes.getRange({
+ start: [channelSlug, pageCount],
+ end: [channelSlug, Number.MAX_SAFE_INTEGER],
+ })) {
+ pageHashes.remove(key as PageHashKey);
+ }
+
+ const channelManifest: ChannelTranscriptsManifest = {
+ version: TRANSCRIPTS_MANIFEST_VERSION,
+ channelSlug,
+ pageCount,
+ maxPageBytes: maxTranscriptPageBytes,
+ generatedAt,
+ slugToPage,
+ };
+ await writeJsonAtomic(
+ path.join(channelDir, "manifest.json"),
+ channelManifest,
+ );
+ }
+
+ await pageHashes.flushed;
+
+ // Prune channel directories for channels no longer in config, and any
+ // legacy per-video `<id>.json` files that leaked in from pre-schema-4
+ // builds at the top level of transcripts/ (pre-channel-subdir layout).
+ const topEntries = await readdir(TRANSCRIPTS_DIR, {
+ withFileTypes: true,
+ }).catch(() => [] as Dirent[]);
+ for (const e of topEntries) {
+ if (e.isDirectory()) {
+ if (!channelConfigs.has(e.name)) {
+ await rm(path.join(TRANSCRIPTS_DIR, e.name), {
+ recursive: true,
+ force: true,
+ });
+ }
+ } else if (e.isFile()) {
+ await rm(path.join(TRANSCRIPTS_DIR, e.name), { force: true });
+ }
+ }
+
+ console.log(
+ `Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`,
+ );
+
// Stream LMDB (reverse composite-key order = newest-date first) to emit
// paginated summaries.
const pageSize = SUMMARIES_PAGE_SIZE;