Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d5cb49794a3ebc589dbcc2bc70409bc0d0e9a7f3
parent 552a70c17bb05455caa3e21be7dd7e89eff79343
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 21 Apr 2026 19:16:14 -0400

Use pages for transcripts

Diffstat:
Mapp/transcriptCache.ts | 74+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Mlib/manifest.ts | 15+++++++++++++++
Mlib/settings.ts | 20+++++++++++++++++++-
Mlib/transcripts.ts | 2++
Mscripts/build-index.ts | 254++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
5 files changed, 344 insertions(+), 21 deletions(-)

diff --git a/app/transcriptCache.ts b/app/transcriptCache.ts @@ -1,10 +1,17 @@ "use client"; import type { TranscriptDetail } from "@/lib/transcripts"; +import type { ChannelTranscriptsManifest } from "@/lib/manifest"; +import { transcriptPageFileName } from "@/lib/manifest"; import { idbGet, idbPutBatch } from "./transcriptStore"; const resolved = new Map<string, TranscriptDetail>(); const inFlight = new Map<string, Promise<TranscriptDetail>>(); +const channelManifests = new Map< + string, + Promise<ChannelTranscriptsManifest> +>(); +const pagePromises = new Map<string, Promise<TranscriptDetail[]>>(); export function fetchTranscript(slug: string): Promise<TranscriptDetail> { const hit = resolved.get(slug); @@ -24,12 +31,69 @@ export function fetchTranscript(slug: string): Promise<TranscriptDetail> { return p; } +function fetchChannelManifest( + channelSlug: string, +): Promise<ChannelTranscriptsManifest> { + let p = channelManifests.get(channelSlug); + if (!p) { + p = fetch(`/transcripts/${channelSlug}/manifest.json`).then((r) => { + if (!r.ok) + throw new Error( + `Failed to fetch transcripts manifest for ${channelSlug}`, + ); + return r.json() as Promise<ChannelTranscriptsManifest>; + }); + p.catch(() => channelManifests.delete(channelSlug)); + channelManifests.set(channelSlug, p); + } + return p; +} + +function fetchPage( + channelSlug: string, + pageIndex: number, +): Promise<TranscriptDetail[]> { + const key = `${channelSlug}:${pageIndex}`; + let p = pagePromises.get(key); + if (!p) { + p = fetch( + `/transcripts/${channelSlug}/${transcriptPageFileName(pageIndex)}`, + ).then((r) => { + if (!r.ok) + throw new Error( + `Failed to fetch transcript page ${channelSlug}/${pageIndex}`, + ); + return r.json() as Promise<TranscriptDetail[]>; + }); + p.catch(() => pagePromises.delete(key)); + pagePromises.set(key, p); + } + return p; +} + async function load(slug: string): Promise<TranscriptDetail> { const stored = await idbGet(slug); if (stored) return stored; - const r = await fetch(`/transcripts/${slug}.json`); - if (!r.ok) throw new Error(`Failed to fetch transcript ${slug}`); - const detail = (await r.json()) as TranscriptDetail; - idbPutBatch(detail); - return detail; + const slashIdx = slug.indexOf("/"); + if (slashIdx < 0) throw new Error(`Malformed transcript slug: ${slug}`); + const channelSlug = slug.slice(0, slashIdx); + const videoId = slug.slice(slashIdx + 1); + const manifest = await fetchChannelManifest(channelSlug); + const pageIndex = manifest.slugToPage[videoId]; + if (pageIndex === undefined) + throw new Error(`Unknown transcript slug: ${slug}`); + const page = await fetchPage(channelSlug, pageIndex); + let found: TranscriptDetail | undefined; + for (const entry of page) { + if (entry.slug === slug) found = entry; + // Opportunistically warm the per-slug memory + IDB caches so subsequent + // fetches for other videos in this page hit without a network round-trip. + // Huge win for searchPipeline which iterates many slugs from the same + // channel/page. + resolved.set(entry.slug, entry); + idbPutBatch(entry); + } + if (!found) + throw new Error(`Transcript ${slug} missing from page ${pageIndex}`); + return found; } diff --git a/lib/manifest.ts b/lib/manifest.ts @@ -18,3 +18,18 @@ export const SUMMARIES_PAGE_SIZE = 1000; export function pageFileName(index: number): string { return `page-${String(index).padStart(4, "0")}.json`; } + +export const TRANSCRIPTS_MANIFEST_VERSION = 1; + +export function transcriptPageFileName(index: number): string { + return `page-${String(index).padStart(4, "0")}.json`; +} + +export type ChannelTranscriptsManifest = { + version: number; + channelSlug: string; + pageCount: number; + maxPageBytes: number; + generatedAt: string; + slugToPage: Record<string, number>; +}; diff --git a/lib/settings.ts b/lib/settings.ts @@ -6,13 +6,19 @@ export type SiteSettings = { siteDescription: string; headerTitle: string; homeTagline: string; + maxTranscriptPageBytes: number; }; +export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; +export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; +export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; + const DEFAULTS: SiteSettings = { siteTitle: "Transcript Browser", siteDescription: "Browse and search video transcripts", headerTitle: "Transcript Browser", homeTagline: "", + maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES, }; let cached: SiteSettings | null = null; @@ -26,6 +32,18 @@ export function getSettings(): SiteSettings { } catch { parsed = {}; } - cached = { ...DEFAULTS, ...parsed }; + const merged = { ...DEFAULTS, ...parsed }; + merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes); + cached = merged; return cached; } + +function clampPageBytes(value: unknown): number { + const n = + typeof value === "number" && Number.isFinite(value) + ? value + : TRANSCRIPT_PAGE_DEFAULT_BYTES; + if (n < TRANSCRIPT_PAGE_MIN_BYTES) return TRANSCRIPT_PAGE_MIN_BYTES; + if (n > TRANSCRIPT_PAGE_HARD_CAP_BYTES) return TRANSCRIPT_PAGE_HARD_CAP_BYTES; + return Math.floor(n); +} diff --git a/lib/transcripts.ts b/lib/transcripts.ts @@ -45,6 +45,8 @@ export type TranscriptDetail = TranscriptSummary & { cues: Cue[] | undefined; }; +export type TranscriptPage = TranscriptDetail[]; + export async function countTranscripts(): Promise<number> { const raw = await readFile(MANIFEST_PATH, "utf8"); const parsed = JSON.parse(raw) as { totalCount?: number }; diff --git a/scripts/build-index.ts b/scripts/build-index.ts @@ -1,13 +1,17 @@ #!/usr/bin/env tsx // Preprocess transcripts/channels/<channelSlug>/data/<videoDir>/ into: // - LMDB cache at transcripts/index.mdb (for incremental rebuilds) -// - public/summaries/{manifest,page-NNNN}.json (paginated summaries index) -// - public/transcripts/<channelSlug>/<id>.json (per-transcript cues + metadata) +// - public/summaries/{manifest,page-NNNN}.json (paginated cross-channel +// summaries index for browse/search) +// - public/transcripts/<channelSlug>/{manifest,page-NNNN}.json +// (per-channel paginated transcript detail + cues, oldest-first so +// adding a newer video only dirties the last page) // // Per-channel config.json selects the transcript parser ("youtube" → VTT, // "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match. import path from "node:path"; +import { createHash } from "node:crypto"; import { mkdir, readdir, @@ -26,13 +30,24 @@ import { toDisplaySummary, type RawMetadata, } from "../lib/transcripts-server"; -import type { TranscriptSummary, DisplaySummary } from "../lib/transcripts"; -import type { Manifest, ChannelEntry } from "../lib/manifest"; +import type { + TranscriptSummary, + TranscriptDetail, + DisplaySummary, +} from "../lib/transcripts"; +import type { + Manifest, + ChannelEntry, + ChannelTranscriptsManifest, +} from "../lib/manifest"; import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE, + TRANSCRIPTS_MANIFEST_VERSION, pageFileName, + transcriptPageFileName, } from "../lib/manifest"; +import { getSettings } from "../lib/settings"; const ROOT = process.cwd(); const CHANNELS_DIR = path.join(ROOT, "transcripts", "channels"); @@ -42,7 +57,7 @@ const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries"); const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts"); const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json"); -const SCHEMA_VERSION = 3; +const SCHEMA_VERSION = 4; type Handling = "youtube" | "transcribe"; type ChannelConfig = { handling: Handling; name?: string }; @@ -50,6 +65,10 @@ type ChannelConfig = { handling: Handling; name?: string }; // Primary sort/lookup key: [uploadDate, channelSlug, id]. Reverse iteration // yields newest-first directly (uploadDate is YYYYMMDD). type IndexKey = [string, string, string]; +// Per-channel forward-scan index: [channelSlug, uploadDate, id]. +type ChannelKey = [string, string, string]; +// Page-hash key: [channelSlug, pageIndex]. +type PageHashKey = [string, number]; // Path key: [channelSlug, videoDir]. Stable regardless of metadata contents, // so diffing by mtime doesn't require parsing metadata.info.json. type PathKey = [string, string]; @@ -60,6 +79,16 @@ type MtimeRecord = { indexKey: IndexKey; }; +type PageHashRecord = { + hash: string; + entryCount: number; + sizeBytes: number; +}; + +function indexToChannelKey(k: IndexKey): ChannelKey { + return [k[1], k[0], k[2]]; +} + type LiveEntry = { channelSlug: string; handling: Handling; @@ -192,6 +221,20 @@ async function main(): Promise<void> { name: "mtimes", encoding: "msgpack", }); + // Per-channel forward index: lets us stream [channelSlug, uploadDate, id] + // in oldest-first order for size-based page packing without loading every + // channel's entries into memory. + const byChannel = root.openDB<number, ChannelKey>({ + name: "byChannel", + encoding: "msgpack", + }); + // Content hashes of emitted transcript pages, so unchanged pages skip the + // write. Since page 1 = oldest videos, adding a new newest-upload video + // only dirties the last page's hash. + const pageHashes = root.openDB<PageHashRecord, PageHashKey>({ + name: "pageHashes", + encoding: "msgpack", + }); const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack", @@ -206,6 +249,8 @@ async function main(): Promise<void> { await sums.clearAsync(); await cues.clearAsync(); await mtimes.clearAsync(); + await byChannel.clearAsync(); + await pageHashes.clearAsync(); await meta.put("schema", SCHEMA_VERSION); } @@ -245,7 +290,9 @@ async function main(): Promise<void> { const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; - // Short-circuit: nothing changed AND public/ is intact. + // Short-circuit: nothing changed AND public/ is intact AND the configured + // max page size matches what each channel was last paginated with. + const { maxTranscriptPageBytes: configuredMaxPageBytes } = getSettings(); if (!anyMutations && !schemaBumped) { const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null); if (manifestRaw) { @@ -261,7 +308,37 @@ async function main(): Promise<void> { SUMMARIES_DIR, pageFileName(Math.max(0, parsed.pageCount - 1)), ); - if ((await exists(firstPage)) && (await exists(lastPage))) { + let transcriptsIntact = true; + for (const channelSlug of channelConfigs.keys()) { + const mPath = path.join( + TRANSCRIPTS_DIR, + channelSlug, + "manifest.json", + ); + const raw = await readFile(mPath, "utf8").catch(() => null); + if (!raw) { + transcriptsIntact = false; + break; + } + try { + const cm = JSON.parse(raw) as ChannelTranscriptsManifest; + if ( + cm.version !== TRANSCRIPTS_MANIFEST_VERSION || + cm.maxPageBytes !== configuredMaxPageBytes + ) { + transcriptsIntact = false; + break; + } + } catch { + transcriptsIntact = false; + break; + } + } + if ( + transcriptsIntact && + (await exists(firstPage)) && + (await exists(lastPage)) + ) { console.log( `Index up to date (${live.length} transcripts, ${parsed.pageCount} pages). Skipping.`, ); @@ -330,22 +407,18 @@ async function main(): Promise<void> { if (prev && !indexKeysEqual(prev.indexKey, indexKey)) { sums.remove(prev.indexKey); cues.remove(prev.indexKey); + byChannel.remove(indexToChannelKey(prev.indexKey)); } sums.put(indexKey, summary); if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); + byChannel.put(indexToChannelKey(indexKey), 1); mtimes.put(pk, { metaMs: s.metaMs, transcriptMs: s.transcriptMs, indexKey, }); - - const detail = { ...summary, cues: cueList }; - await writeJsonAtomic( - path.join(TRANSCRIPTS_DIR, `${summary.slug}.json`), - detail, - ); } catch (err) { console.warn( `Failed to process ${s.channelSlug}/${s.videoDir}:`, @@ -365,15 +438,166 @@ async function main(): Promise<void> { for (const { pathKey, indexKey } of removed) { sums.remove(indexKey); cues.remove(indexKey); + byChannel.remove(indexToChannelKey(indexKey)); mtimes.remove(pathKey); - const slug = `${indexKey[1]}/${indexKey[2]}`; - await rm(path.join(TRANSCRIPTS_DIR, `${slug}.json`), { force: true }); } await sums.flushed; await cues.flushed; + await byChannel.flushed; await mtimes.flushed; + // Emit per-channel paginated transcript pages. Pages are size-packed + // oldest-first, so newer videos land on the last page and older pages + // hash-match from build to build. + const maxTranscriptPageBytes = configuredMaxPageBytes; + const generatedAt = new Date().toISOString(); + let pagesWritten = 0; + let pagesSkipped = 0; + let pagesDeleted = 0; + + type PendingEntry = { encoded: string; id: string }; + + const writePage = async ( + channelSlug: string, + idx: number, + entries: PendingEntry[], + ): Promise<void> => { + const body = `[${entries.map((e) => e.encoded).join(",")}]`; + const hash = createHash("sha1").update(body).digest("hex"); + const prev = pageHashes.get([channelSlug, idx]); + const outPath = path.join( + TRANSCRIPTS_DIR, + channelSlug, + transcriptPageFileName(idx), + ); + if (prev?.hash === hash && (await exists(outPath))) { + pagesSkipped++; + return; + } + const sizeBytes = Buffer.byteLength(body, "utf8"); + const tmp = `${outPath}.tmp-${process.pid}`; + await writeFile(tmp, body); + await rename(tmp, outPath); + pageHashes.put([channelSlug, idx], { + hash, + entryCount: entries.length, + sizeBytes, + }); + pagesWritten++; + }; + + for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + const channelDir = path.join(TRANSCRIPTS_DIR, channelSlug); + await mkdir(channelDir, { recursive: true }); + + let pageIdx = 0; + let buffer: PendingEntry[] = []; + // Running payload size excluding the outer `[` and `]` (accounted at + // flush time). Includes the leading commas between elements. + let payloadBytes = 0; + const slugToPage: Record<string, number> = {}; + + for (const { key } of byChannel.getRange({ + start: [channelSlug], + end: [channelSlug, "￿"], + })) { + const ck = key as ChannelKey; + if (ck[0] !== channelSlug) continue; + const indexKey: IndexKey = [ck[1], ck[0], ck[2]]; + const summary = sums.get(indexKey); + if (!summary) continue; + const cueList = cues.get(indexKey); + const detail: TranscriptDetail = { ...summary, cues: cueList }; + const encoded = JSON.stringify(detail); + const entryBytes = Buffer.byteLength(encoded, "utf8"); + const commaBytes = buffer.length === 0 ? 0 : 1; + const delta = entryBytes + commaBytes; + + // Close current page if this entry would push it past the limit. + // `+ 2` accounts for the outer brackets. + if (buffer.length > 0 && payloadBytes + delta + 2 > maxTranscriptPageBytes) { + await writePage(channelSlug, pageIdx, buffer); + pageIdx++; + buffer = []; + payloadBytes = 0; + } + + if (entryBytes + 2 > maxTranscriptPageBytes) { + console.warn( + ` ${summary.slug}: entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`, + ); + } + + slugToPage[summary.id] = pageIdx; + buffer.push({ encoded, id: summary.id }); + payloadBytes += delta; + } + + if (buffer.length > 0) { + await writePage(channelSlug, pageIdx, buffer); + pageIdx++; + } + + const pageCount = pageIdx; + + // Prune stale per-channel page files + drop pageHashes beyond pageCount. + const keep = new Set<string>(["manifest.json"]); + for (let i = 0; i < pageCount; i++) keep.add(transcriptPageFileName(i)); + const existing = await readdir(channelDir).catch(() => [] as string[]); + for (const name of existing) { + if (keep.has(name)) continue; + await rm(path.join(channelDir, name), { force: true }); + pagesDeleted++; + } + // pageHashes may still have records for page indexes >= pageCount + // (channel shrank). Drop them. + for (const { key } of pageHashes.getRange({ + start: [channelSlug, pageCount], + end: [channelSlug, Number.MAX_SAFE_INTEGER], + })) { + pageHashes.remove(key as PageHashKey); + } + + const channelManifest: ChannelTranscriptsManifest = { + version: TRANSCRIPTS_MANIFEST_VERSION, + channelSlug, + pageCount, + maxPageBytes: maxTranscriptPageBytes, + generatedAt, + slugToPage, + }; + await writeJsonAtomic( + path.join(channelDir, "manifest.json"), + channelManifest, + ); + } + + await pageHashes.flushed; + + // Prune channel directories for channels no longer in config, and any + // legacy per-video `<id>.json` files that leaked in from pre-schema-4 + // builds at the top level of transcripts/ (pre-channel-subdir layout). + const topEntries = await readdir(TRANSCRIPTS_DIR, { + withFileTypes: true, + }).catch(() => [] as Dirent[]); + for (const e of topEntries) { + if (e.isDirectory()) { + if (!channelConfigs.has(e.name)) { + await rm(path.join(TRANSCRIPTS_DIR, e.name), { + recursive: true, + force: true, + }); + } + } else if (e.isFile()) { + await rm(path.join(TRANSCRIPTS_DIR, e.name), { force: true }); + } + } + + console.log( + `Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`, + ); + // Stream LMDB (reverse composite-key order = newest-date first) to emit // paginated summaries. const pageSize = SUMMARIES_PAGE_SIZE;