Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 96f61d843407b8b2a56c9a137fffa6a7bac9716f
parent 103b21f4e2e2bd5eaf081d2d05eec6f8f0e1533a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 18 Apr 2026 00:58:41 -0400

overhaul for 10k+ vids

Diffstat:
M.gitignore | 5+++++
Mapp/PlayerProvider.tsx | 8++++++++
Aapp/QueryProvider.tsx | 23+++++++++++++++++++++++
Mapp/TranscriptModal.tsx | 11++++++++---
Mapp/TranscriptSearch.tsx | 424+++++++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Aapp/badges.tsx | 24++++++++++++++++++++++++
Mapp/layout.tsx | 23+++++++++++++----------
Aapp/searchPipeline.ts | 244+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Dapp/summaries/route.ts | 7-------
Mapp/summariesCache.ts | 75++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Mapp/transcriptCache.ts | 36+++++++++++++++++++++++++++---------
Dapp/transcripts/[slug]/route.ts | 15---------------
Mapp/urlState.ts | 39+++++++++++++++++++++++++++++++++++++--
Alib/manifest.ts | 20++++++++++++++++++++
Alib/transcripts-server.ts | 52++++++++++++++++++++++++++++++++++++++++++++++++++++
Mlib/transcripts.ts | 97+++++++++++++++----------------------------------------------------------------
Mnext.config.ts | 1+
Mpackage.json | 4++++
Mpnpm-lock.yaml | 197+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Ascripts/build-index.ts | 299+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aserve.json | 14++++++++++++++
21 files changed, 1335 insertions(+), 283 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -48,3 +48,8 @@ next-env.d.ts /public/code.tar.xz /public/transcripts.tar.gz /public/transcripts.tar.xz + +# preprocessed transcript data (regenerate with `pnpm build:index`) +/public/summaries/ +/public/transcripts/ +/transcripts/index.mdb/ diff --git a/app/PlayerProvider.tsx b/app/PlayerProvider.tsx @@ -30,6 +30,8 @@ export type TranscriptData = { duration: string; channel: string; description?: string; + isLivestream: boolean; + ageRestricted: boolean; cues?: Cue[]; }; @@ -44,6 +46,8 @@ type Detail = { duration: number; channel: string; description?: string; + isLivestream: boolean; + ageRestricted: boolean; cues?: Cue[]; }; @@ -127,6 +131,8 @@ export function PlayerProvider({ duration: formatDuration(detail.duration), channel: detail.channel, description: detail.description, + isLivestream: detail.isLivestream, + ageRestricted: detail.ageRestricted, cues: detail.cues, }; }, [detail, detailMatches]); @@ -228,6 +234,8 @@ export function PlayerProvider({ duration: full.duration, channel: full.channel, description: full.description, + isLivestream: full.isLivestream, + ageRestricted: full.ageRestricted, cues: full.cues, }); }) diff --git a/app/QueryProvider.tsx b/app/QueryProvider.tsx @@ -0,0 +1,23 @@ +"use client"; + +import { useState } from "react"; +import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; + +export function QueryProvider({ children }: { children: React.ReactNode }) { + const [client] = useState( + () => + new QueryClient({ + defaultOptions: { + queries: { + // Summaries and manifest are immutable across a deploy; never + // consider them stale so TQ won't refetch on focus/mount. + staleTime: Infinity, + gcTime: Infinity, + refetchOnWindowFocus: false, + retry: 1, + }, + }, + }), + ); + return <QueryClientProvider client={client}>{children}</QueryClientProvider>; +} diff --git a/app/TranscriptModal.tsx b/app/TranscriptModal.tsx @@ -2,6 +2,7 @@ import { useEffect, useRef, useState } from "react"; import { ControlButton, toHMS, usePlayer } from "./PlayerProvider"; +import { AgeRestrictedBadge, LivestreamBadge } from "./badges"; import { formatTimestamp } from "@/lib/vtt"; import { formatDate } from "@/lib/format"; @@ -132,9 +133,13 @@ export default function TranscriptModal() { {data?.title ?? "Loading…"} </h2> {data && ( - <p className="text-xs text-zinc-300 mt-0.5"> - {formatDate(data.uploadDate)} - {data.channel && ` · ${data.channel}`} + <p className="text-xs text-zinc-300 mt-0.5 flex items-center gap-1.5 flex-wrap"> + <span> + {formatDate(data.uploadDate)} + {data.channel && ` · ${data.channel}`} + </span> + {data.isLivestream && <LivestreamBadge />} + {data.ageRestricted && <AgeRestrictedBadge />} </p> )} </div> diff --git a/app/TranscriptSearch.tsx b/app/TranscriptSearch.tsx @@ -1,28 +1,34 @@ "use client"; -import { useEffect, useMemo, useState } from "react"; +import { useEffect, useMemo, useRef, useState } from "react"; import { usePlayer } from "./PlayerProvider"; -import { fetchTranscript } from "./transcriptCache"; -import { fetchSummaries } from "./summariesCache"; +import { useSummaries } from "./summariesCache"; +import { + createSearchPipeline, + filterSlugs, + type Hit, + type PipelineController, +} from "./searchPipeline"; import { useUrlParams, writeUrlParams } from "./urlState"; -import type { DisplaySummary, TranscriptDetail } from "@/lib/transcripts"; +import { AgeRestrictedBadge, LivestreamBadge } from "./badges"; +import type { DisplaySummary } from "@/lib/transcripts"; type Summary = DisplaySummary; -type Hit = { - start: number; - text: string; -}; - type HitGroup = { slug: string; title: string; channel: string; date: string; + isLivestream: boolean; + ageRestricted: boolean; hits: Hit[]; }; -const MAX_HITS = 1000; +const MAX_HITS = 500; +const FETCH_CONCURRENCY = 6; +const FLUSH_INTERVAL_MS = 120; +const GROUPS_PAGE_SIZE = 50; export default function TranscriptSearch() { const { @@ -30,68 +36,98 @@ export default function TranscriptSearch() { re: useRegex, v: activeVideo, t: activeTime, + ch: committedChannelList, + nov, + nol, + naa, + nar, } = useUrlParams(); + // Content-based key so Set identity doesn't churn when other URL params + // (v/t) change — otherwise opening the player modal would restart the + // search pipeline from scratch. + const committedChannelsKey = committedChannelList + .slice() + .sort() + .join("\u0000"); + const committedChannels = useMemo( + () => new Set(committedChannelList), + // eslint-disable-next-line react-hooks/exhaustive-deps + [committedChannelsKey], + ); const [input, setInput] = useState(query); useEffect(() => { setInput(query); }, [query]); + + const [draftExcludedChannels, setDraftExcludedChannels] = useState< + Set<string> + >(() => new Set(committedChannelList)); + const [draftNov, setDraftNov] = useState(nov); + const [draftNol, setDraftNol] = useState(nol); + const [draftNaa, setDraftNaa] = useState(naa); + const [draftNar, setDraftNar] = useState(nar); + useEffect(() => { + setDraftExcludedChannels(new Set(committedChannelList)); + }, [committedChannelList]); + useEffect(() => setDraftNov(nov), [nov]); + useEffect(() => setDraftNol(nol), [nol]); + useEffect(() => setDraftNaa(naa), [naa]); + useEffect(() => setDraftNar(nar), [nar]); + const [searchExecuted, setSearchExecuted] = useState(() => { if (typeof window === "undefined") return true; const params = new URLSearchParams(window.location.search); return !(params.has("v") && (params.get("q") ?? "") !== ""); }); - const [transcripts, setTranscripts] = useState<Summary[] | null>(null); - const [fetched, setFetched] = useState<Record<string, TranscriptDetail>>({}); + const { manifest, summaries, loadedPages, pageCount, summariesReady } = + useSummaries(); + const transcripts: Summary[] | null = summariesReady ? summaries : null; + + const [hitsBySlug, setHitsBySlug] = useState<Record<string, Hit[]>>({}); + const [totalHits, setTotalHits] = useState(0); + const [processed, setProcessed] = useState(0); + const [totalToProcess, setTotalToProcess] = useState(0); + const [pipelineDone, setPipelineDone] = useState(false); + const [capped, setCapped] = useState(false); const [hitLimit, setHitLimit] = useState(MAX_HITS); + const [groupsShown, setGroupsShown] = useState(GROUPS_PAGE_SIZE); + const pipelineRef = useRef<PipelineController | null>(null); const { openTranscript } = usePlayer(); const trimmed = searchExecuted ? query.trim() : ""; useEffect(() => { setHitLimit(MAX_HITS); + setGroupsShown(GROUPS_PAGE_SIZE); }, [trimmed, useRegex]); - useEffect(() => { - if (!trimmed) return; - if (transcripts) return; - let cancelled = false; - fetchSummaries() - .then((list) => { - if (cancelled) return; - setTranscripts(list); - }) - .catch(() => {}); - return () => { - cancelled = true; - }; - }, [trimmed, transcripts]); - - useEffect(() => { - if (!trimmed) return; - if (!transcripts) return; - let cancelled = false; - for (const t of transcripts) { - if (fetched[t.slug]) continue; - fetchTranscript(t.slug) - .then((full) => { - if (cancelled) return; - setFetched((prev) => - prev[t.slug] ? prev : { ...prev, [t.slug]: full }, - ); - }) - .catch(() => {}); + const channelOptions = useMemo<string[]>(() => { + if (manifest?.channels.length) { + return manifest.channels + .map((c) => c.name) + .filter((n) => n) + .sort((a, b) => a.localeCompare(b)); } - return () => { - cancelled = true; + if (!transcripts) return []; + const names = new Set<string>(); + for (const t of transcripts) if (t.channel) names.add(t.channel); + return Array.from(names).sort((a, b) => a.localeCompare(b)); + }, [manifest, transcripts]); + + const passesFilter = useMemo(() => { + return (t: Summary) => { + if (committedChannels.has(t.channel)) return false; + if (t.isLivestream ? nol : nov) return false; + if (t.ageRestricted ? nar : naa) return false; + return true; }; - }, [trimmed, transcripts, fetched]); + }, [committedChannels, nov, nol, naa, nar]); - const loadedCount = Object.keys(fetched).length; - const totalCount = transcripts?.length ?? 0; - const indexLoading = Boolean(trimmed) && transcripts === null; - const fetching = - Boolean(trimmed) && transcripts !== null && loadedCount < totalCount; + // Stable signature of the filter inputs; used as a single dep in the search + // pipeline so reference-unstable upstream values (new `ch: []` array on every + // URL change) don't cancel an in-flight search when the modal opens. + const filterKey = `${committedChannelsKey}|${nov ? 1 : 0}|${nol ? 1 : 0}|${naa ? 1 : 0}|${nar ? 1 : 0}`; const regex = useMemo<{ re: RegExp | null; error: string | null }>(() => { if (!trimmed || !useRegex) return { re: null, error: null }; @@ -102,71 +138,111 @@ export default function TranscriptSearch() { } }, [trimmed, useRegex]); - const { groups, totalHits, capped } = useMemo<{ - groups: HitGroup[]; - totalHits: number; - capped: boolean; - }>(() => { - if (!trimmed) return { groups: [], totalHits: 0, capped: false }; - if (!transcripts) return { groups: [], totalHits: 0, capped: false }; - if (useRegex && !regex.re) - return { groups: [], totalHits: 0, capped: false }; - const groups: HitGroup[] = []; - let total = 0; + // Pipeline lifecycle: create a new pipeline when inputs change, and cancel + // the prior one. hitLimit is NOT a dep here — it's pushed into the live + // pipeline via setHitLimit so bumping the cap resumes rather than restarts. + useEffect(() => { + pipelineRef.current?.cancel(); + pipelineRef.current = null; + + if (!trimmed) { + setHitsBySlug({}); + setTotalHits(0); + setProcessed(0); + setTotalToProcess(0); + setPipelineDone(false); + setCapped(false); + return; + } + if (!transcripts) return; + if (useRegex && !regex.re) return; + + const slugs = filterSlugs(transcripts, passesFilter); + + const controller = createSearchPipeline({ + slugs, + query: trimmed, + useRegex, + regex: regex.re, + initialHitLimit: hitLimit, + concurrency: FETCH_CONCURRENCY, + flushIntervalMs: FLUSH_INTERVAL_MS, + emit: (u) => { + setHitsBySlug(u.hitsBySlug); + setTotalHits(u.totalHits); + setProcessed(u.processed); + setTotalToProcess(u.totalToProcess); + setCapped(u.capped); + setPipelineDone(u.done); + }, + }); + pipelineRef.current = controller; + + return () => { + controller.cancel(); + if (pipelineRef.current === controller) pipelineRef.current = null; + }; + // hitLimit intentionally omitted — bumped via setHitLimit below. + // passesFilter captured via filterKey (content signature). + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [trimmed, transcripts, filterKey, useRegex, regex.re]); + + // Push hitLimit bumps into the live pipeline so "Show more videos" resumes + // the existing traversal instead of restarting from slug 0. + useEffect(() => { + pipelineRef.current?.setHitLimit(hitLimit); + }, [hitLimit]); + + const groups = useMemo<HitGroup[]>(() => { + if (!trimmed || !transcripts) return []; + const out: HitGroup[] = []; for (const t of transcripts) { - const full = fetched[t.slug]; - if (!full?.cues) continue; - const cues = full.cues; - const hits: Hit[] = []; - for (let i = 0; i < cues.length; i++) { - const cur = cues[i]; - const prevText = i > 0 ? cues[i - 1].text : ""; - const nextText = i < cues.length - 1 ? cues[i + 1].text : ""; - const sep1 = prevText ? " " : ""; - const sep2 = nextText ? " " : ""; - const windowText = prevText + sep1 + cur.text + sep2 + nextText; - const curStart = prevText.length + sep1.length; - const curEnd = curStart + cur.text.length; - const m = findFirstMatchInRange( - windowText, - curStart, - curEnd, - trimmed, - useRegex, - regex.re, - ); - if (!m) continue; - const crosses = m.idx < curStart || m.idx + m.length > curEnd; - hits.push({ - start: Math.round(cur.start), - text: crosses ? windowText : cur.text, - }); - total++; - if (total >= hitLimit) break; - } - if (hits.length > 0) { - groups.push({ - slug: t.slug, - title: t.title, - channel: t.channel, - date: t.date, - hits, - }); - } - if (total >= hitLimit) { - return { groups, totalHits: total, capped: true }; - } + const hits = hitsBySlug[t.slug]; + if (!hits?.length) continue; + out.push({ + slug: t.slug, + title: t.title, + channel: t.channel, + date: t.date, + isLivestream: t.isLivestream, + ageRestricted: t.ageRestricted, + hits, + }); } - return { groups, totalHits: total, capped: false }; - }, [trimmed, transcripts, fetched, useRegex, regex.re, hitLimit]); + return out; + }, [trimmed, transcripts, hitsBySlug]); + + const indexLoading = Boolean(trimmed) && !summariesReady; + const searching = Boolean(trimmed) && totalToProcess > 0; + const pipelineActive = searching && !pipelineDone; + const indexProgress = + pageCount > 0 ? `${loadedPages}/${pageCount} pages` : ""; + + const filtersDirty = + !sameSet(draftExcludedChannels, committedChannels) || + draftNov !== nov || + draftNol !== nol || + draftNaa !== naa || + draftNar !== nar; + const queryDirty = input.trim() !== (searchExecuted ? query.trim() : ""); + const commitSearch = () => { + setSearchExecuted(true); + writeUrlParams({ + q: input.trim(), + ch: Array.from(draftExcludedChannels), + nov: draftNov, + nol: draftNol, + naa: draftNaa, + nar: draftNar, + }); + }; return ( <div className="flex flex-col gap-6"> <form onSubmit={(e) => { e.preventDefault(); - setSearchExecuted(true); - writeUrlParams({ q: input.trim() }); + commitSearch(); }} className="flex flex-col gap-1" > @@ -200,7 +276,7 @@ export default function TranscriptSearch() { {regex.error} </span> )} - {input.trim() !== (searchExecuted ? query.trim() : "") && ( + {(queryDirty || filtersDirty) && ( <span className="text-xs text-zinc-400 ml-auto"> Press Enter to search </span> @@ -208,6 +284,78 @@ export default function TranscriptSearch() { </label> </form> + {channelOptions.length > 0 && ( + <div className="flex flex-wrap items-center gap-x-4 gap-y-2 text-sm text-zinc-600 dark:text-zinc-400 -mt-3"> + {channelOptions.length > 1 && ( + <div className="flex flex-wrap items-center gap-x-3 gap-y-1"> + <span className="text-xs uppercase tracking-wide text-zinc-500"> + Channels + </span> + {channelOptions.map((name) => { + const checked = !draftExcludedChannels.has(name); + return ( + <label + key={name} + className="flex items-center gap-1.5 select-none" + > + <input + type="checkbox" + checked={checked} + onChange={(e) => { + setDraftExcludedChannels((prev) => { + const next = new Set(prev); + if (e.target.checked) next.delete(name); + else next.add(name); + return next; + }); + }} + className="accent-blue-600" + /> + {name} + </label> + ); + })} + </div> + )} + <label className="flex items-center gap-1.5 select-none"> + <input + type="checkbox" + checked={!draftNov} + onChange={(e) => setDraftNov(!e.target.checked)} + className="accent-blue-600" + /> + Videos + </label> + <label className="flex items-center gap-1.5 select-none"> + <input + type="checkbox" + checked={!draftNol} + onChange={(e) => setDraftNol(!e.target.checked)} + className="accent-blue-600" + /> + Livestreams + </label> + <label className="flex items-center gap-1.5 select-none"> + <input + type="checkbox" + checked={!draftNaa} + onChange={(e) => setDraftNaa(!e.target.checked)} + className="accent-blue-600" + /> + All ages + </label> + <label className="flex items-center gap-1.5 select-none"> + <input + type="checkbox" + checked={!draftNar} + onChange={(e) => setDraftNar(!e.target.checked)} + className="accent-blue-600" + /> + Age-restricted + </label> + </div> + )} + {!trimmed && ( <p className="text-sm text-zinc-500"> Enter a query to search transcripts. @@ -230,21 +378,22 @@ export default function TranscriptSearch() { </span> {indexLoading && ( <span className="text-xs text-zinc-400 font-normal"> - loading index… + loading index{indexProgress ? ` (${indexProgress})` : ""}… </span> )} - {fetching && ( + {searching && ( <span className="text-xs text-zinc-400 font-normal"> - loaded {loadedCount}/{totalCount}… + searched {processed}/{totalToProcess} + {pipelineActive ? "…" : ""} </span> )} </h2> - {!indexLoading && groups.length === 0 && !fetching && ( + {!indexLoading && groups.length === 0 && !pipelineActive && ( <p className="text-sm text-zinc-500">No transcript mentions.</p> )} {groups.length > 0 && ( <div className="flex flex-col gap-3"> - {groups.map((g) => ( + {groups.slice(0, groupsShown).map((g) => ( <div key={g.slug} className="border border-zinc-200 dark:border-zinc-800 rounded-lg overflow-hidden bg-white dark:bg-zinc-900" @@ -257,6 +406,8 @@ export default function TranscriptSearch() { <span className="font-medium truncate flex-1 min-w-0"> {g.title} </span> + {g.isLivestream && <LivestreamBadge />} + {g.ageRestricted && <AgeRestrictedBadge />} <span className="text-xs text-zinc-500 shrink-0"> {g.channel && `${g.channel} · `} {g.date} @@ -293,14 +444,19 @@ export default function TranscriptSearch() { ))} </div> )} - {capped && ( + {(groups.length > groupsShown || capped) && ( <div className="mt-3 flex justify-center"> <button type="button" - onClick={() => setHitLimit((l) => l + MAX_HITS)} + onClick={() => { + setGroupsShown((n) => n + GROUPS_PAGE_SIZE); + if (capped) setHitLimit((l) => l + MAX_HITS); + }} className="px-4 py-2 rounded-md border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 hover:bg-zinc-50 dark:hover:bg-zinc-800 text-sm" > - Load more + {groups.length > groupsShown + ? `Show more videos (${groups.length - groupsShown} loaded${capped ? ", more available" : ""})` + : "Show more videos"} </button> </div> )} @@ -311,40 +467,10 @@ export default function TranscriptSearch() { ); } -function findFirstMatchInRange( - haystack: string, - rangeStart: number, - rangeEnd: number, - query: string, - useRegex: boolean, - regex: RegExp | null, -): { idx: number; length: number } | null { - if (useRegex) { - if (!regex) return null; - const flags = regex.flags.includes("g") ? regex.flags : regex.flags + "g"; - const re = new RegExp(regex.source, flags); - let m: RegExpExecArray | null; - while ((m = re.exec(haystack)) !== null) { - if (m.index >= rangeStart && m.index < rangeEnd) { - return { idx: m.index, length: m[0].length }; - } - if (m.index >= rangeEnd) return null; - if (m[0].length === 0) re.lastIndex++; - } - return null; - } - if (!query) return null; - const lower = haystack.toLowerCase(); - const ql = query.toLowerCase(); - let from = 0; - while (from <= haystack.length) { - const idx = lower.indexOf(ql, from); - if (idx === -1) return null; - if (idx >= rangeStart && idx < rangeEnd) return { idx, length: ql.length }; - if (idx >= rangeEnd) return null; - from = idx + 1; - } - return null; +function sameSet<T>(a: Set<T>, b: Set<T>): boolean { + if (a.size !== b.size) return false; + for (const v of a) if (!b.has(v)) return false; + return true; } function formatSeconds(s: number): string { diff --git a/app/badges.tsx b/app/badges.tsx @@ -0,0 +1,24 @@ +const badgeBase = + "shrink-0 text-[10px] font-semibold uppercase tracking-wide px-1.5 py-0.5 rounded"; + +export function LivestreamBadge() { + return ( + <span + title="Livestream" + className={`${badgeBase} bg-zinc-200 text-zinc-700 dark:bg-zinc-700 dark:text-zinc-200`} + > + Live + </span> + ); +} + +export function AgeRestrictedBadge() { + return ( + <span + title="Age-restricted" + className={`${badgeBase} bg-rose-100 text-rose-800 dark:bg-rose-900/40 dark:text-rose-200`} + > + 18+ + </span> + ); +} diff --git a/app/layout.tsx b/app/layout.tsx @@ -2,6 +2,7 @@ import type { Metadata } from "next"; import { Geist, Geist_Mono } from "next/font/google"; import Link from "next/link"; import { getSettings } from "@/lib/settings"; +import { QueryProvider } from "./QueryProvider"; import "./globals.css"; const geistSans = Geist({ @@ -34,16 +35,18 @@ export default async function RootLayout({ className={`${geistSans.variable} ${geistMono.variable} h-full antialiased`} > <body className="min-h-full flex flex-col bg-zinc-50 text-zinc-900 dark:bg-zinc-950 dark:text-zinc-100 font-sans"> - <header className="border-b border-zinc-200 dark:border-zinc-800 bg-white/70 dark:bg-zinc-900/70 backdrop-blur sticky top-0 z-20"> - <div className="max-w-6xl mx-auto px-4 py-3 flex items-center justify-between"> - <Link href="/" className="font-semibold tracking-tight"> - {settings.headerTitle} - </Link> - </div> - </header> - <main className="flex-1 w-full max-w-6xl mx-auto px-4 py-6"> - {children} - </main> + <QueryProvider> + <header className="border-b border-zinc-200 dark:border-zinc-800 bg-white/70 dark:bg-zinc-900/70 backdrop-blur sticky top-0 z-20"> + <div className="max-w-6xl mx-auto px-4 py-3 flex items-center justify-between"> + <Link href="/" className="font-semibold tracking-tight"> + {settings.headerTitle} + </Link> + </div> + </header> + <main className="flex-1 w-full max-w-6xl mx-auto px-4 py-6"> + {children} + </main> + </QueryProvider> </body> </html> ); diff --git a/app/searchPipeline.ts b/app/searchPipeline.ts @@ -0,0 +1,244 @@ +import { fetchTranscript } from "./transcriptCache"; +import type { DisplaySummary } from "@/lib/transcripts"; + +export type Hit = { start: number; text: string }; + +export type PipelineUpdate = { + hitsBySlug: Record<string, Hit[]>; + totalHits: number; + processed: number; + totalToProcess: number; + capped: boolean; + done: boolean; +}; + +type PipelineConfig = { + slugs: string[]; + query: string; + useRegex: boolean; + regex: RegExp | null; + initialHitLimit: number; + concurrency: number; + flushIntervalMs: number; + emit: (update: PipelineUpdate) => void; +}; + +export type PipelineController = { + cancel(): void; + setHitLimit(limit: number): void; +}; + +// Runs a streaming search over the given slugs. State is closure-captured so +// `setHitLimit` can raise the cap and re-spawn workers without restarting +// traversal — workers resume from the shared `idx` cursor. +export function createSearchPipeline( + config: PipelineConfig, +): PipelineController { + const { + slugs, + query, + useRegex, + regex, + initialHitLimit, + concurrency, + flushIntervalMs, + emit, + } = config; + + let cancelled = false; + let idx = 0; + let totalSoFar = 0; + let hitLimit = initialHitLimit; + let activeWorkers = 0; + let done = false; + const localHits: Record<string, Hit[]> = {}; + let flushTimer: number | null = null; + + const pushUpdate = (overrides: Partial<PipelineUpdate> = {}) => { + emit({ + hitsBySlug: { ...localHits }, + totalHits: totalSoFar, + processed: idx, + totalToProcess: slugs.length, + capped: totalSoFar >= hitLimit && idx < slugs.length, + done, + ...overrides, + }); + }; + + const scheduleFlush = () => { + if (flushTimer !== null || cancelled) return; + flushTimer = window.setTimeout(() => { + flushTimer = null; + if (cancelled) return; + pushUpdate(); + }, flushIntervalMs); + }; + + const finalize = () => { + if (done || cancelled) return; + done = true; + if (flushTimer !== null) { + window.clearTimeout(flushTimer); + flushTimer = null; + } + pushUpdate(); + }; + + const worker = async () => { + activeWorkers++; + try { + while (!cancelled) { + if (totalSoFar >= hitLimit) return; + const my = idx++; + if (my >= slugs.length) return; + const slug = slugs[my]; + try { + const full = await fetchTranscript(slug); + if (cancelled) return; + if (totalSoFar >= hitLimit) return; + if (full.cues) { + const remaining = hitLimit - totalSoFar; + const hits = findHitsInCues( + full.cues, + query, + useRegex, + regex, + remaining, + ); + if (hits.length > 0) { + localHits[slug] = hits; + totalSoFar += hits.length; + } + } + } catch { + // ignore per-transcript failures + } + scheduleFlush(); + } + } finally { + activeWorkers--; + if (activeWorkers === 0 && !cancelled) { + // Either we hit the cap or ran out of work — either way, settle. + finalize(); + } + } + }; + + const ensureWorkers = () => { + if (cancelled || done) return; + if (totalSoFar >= hitLimit) return; + if (idx >= slugs.length) return; + const needed = Math.min( + concurrency - activeWorkers, + slugs.length - idx, + ); + for (let i = 0; i < needed; i++) worker(); + }; + + // Emit initial snapshot synchronously so the UI clears previous results. + pushUpdate(); + ensureWorkers(); + + return { + cancel() { + cancelled = true; + if (flushTimer !== null) { + window.clearTimeout(flushTimer); + flushTimer = null; + } + }, + setHitLimit(limit: number) { + if (cancelled) return; + if (limit <= hitLimit) return; + hitLimit = limit; + // The prior run may have finalized because we were capped. Resume. + if (done) { + done = false; + pushUpdate(); + } + ensureWorkers(); + }, + }; +} + +function findHitsInCues( + cues: { start: number; text: string }[], + query: string, + useRegex: boolean, + regex: RegExp | null, + limit: number, +): Hit[] { + const hits: Hit[] = []; + for (let i = 0; i < cues.length && hits.length < limit; i++) { + const cur = cues[i]; + const prevText = i > 0 ? cues[i - 1].text : ""; + const nextText = i < cues.length - 1 ? cues[i + 1].text : ""; + const sep1 = prevText ? " " : ""; + const sep2 = nextText ? " " : ""; + const windowText = prevText + sep1 + cur.text + sep2 + nextText; + const curStart = prevText.length + sep1.length; + const curEnd = curStart + cur.text.length; + const m = findFirstMatchInRange( + windowText, + curStart, + curEnd, + query, + useRegex, + regex, + ); + if (!m) continue; + const crosses = m.idx < curStart || m.idx + m.length > curEnd; + hits.push({ + start: Math.round(cur.start), + text: crosses ? windowText : cur.text, + }); + } + return hits; +} + +function findFirstMatchInRange( + haystack: string, + rangeStart: number, + rangeEnd: number, + query: string, + useRegex: boolean, + regex: RegExp | null, +): { idx: number; length: number } | null { + if (useRegex) { + if (!regex) return null; + const flags = regex.flags.includes("g") ? regex.flags : regex.flags + "g"; + const re = new RegExp(regex.source, flags); + let m: RegExpExecArray | null; + while ((m = re.exec(haystack)) !== null) { + if (m.index >= rangeStart && m.index < rangeEnd) { + return { idx: m.index, length: m[0].length }; + } + if (m.index >= rangeEnd) return null; + if (m[0].length === 0) re.lastIndex++; + } + return null; + } + if (!query) return null; + const lower = haystack.toLowerCase(); + const ql = query.toLowerCase(); + let from = 0; + while (from <= haystack.length) { + const idx = lower.indexOf(ql, from); + if (idx === -1) return null; + if (idx >= rangeStart && idx < rangeEnd) return { idx, length: ql.length }; + if (idx >= rangeEnd) return null; + from = idx + 1; + } + return null; +} + +// Helper to build slug list from summaries + filter predicate. +export function filterSlugs( + summaries: DisplaySummary[], + passes: (t: DisplaySummary) => boolean, +): string[] { + const slugs: string[] = []; + for (const t of summaries) if (passes(t)) slugs.push(t.slug); + return slugs; +} diff --git a/app/summaries/route.ts b/app/summaries/route.ts @@ -1,7 +0,0 @@ -import { listDisplaySummaries } from "@/lib/transcripts"; - -export const dynamic = "force-static"; - -export async function GET() { - return Response.json(await listDisplaySummaries()); -} diff --git a/app/summariesCache.ts b/app/summariesCache.ts @@ -1,17 +1,74 @@ "use client"; +import { useMemo } from "react"; +import { useQueries, useQuery } from "@tanstack/react-query"; import type { DisplaySummary } from "@/lib/transcripts"; +import type { Manifest } from "@/lib/manifest"; +import { pageFileName } from "@/lib/manifest"; -let promise: Promise<DisplaySummary[]> | null = null; +async function fetchJson<T>(url: string): Promise<T> { + const r = await fetch(url); + if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); + return (await r.json()) as T; +} + +export type SummariesState = { + manifest: Manifest | null; + summaries: DisplaySummary[]; + loadedPages: number; + pageCount: number; + summariesReady: boolean; + error: Error | null; +}; -export function fetchSummaries(): Promise<DisplaySummary[]> { - if (promise) return promise; - promise = fetch("/summaries").then((r) => { - if (!r.ok) throw new Error(`Failed to fetch summaries: ${r.status}`); - return r.json() as Promise<DisplaySummary[]>; +export function useManifest() { + return useQuery<Manifest>({ + queryKey: ["manifest"], + queryFn: () => fetchJson<Manifest>("/summaries/manifest.json"), }); - promise.catch(() => { - promise = null; +} + +export function useSummaries(): SummariesState { + const manifestQuery = useManifest(); + const manifest = manifestQuery.data ?? null; + const pageCount = manifest?.pageCount ?? 0; + + const pageQueries = useQueries({ + queries: Array.from({ length: pageCount }, (_, i) => ({ + queryKey: ["summaries-page", i], + queryFn: () => + fetchJson<DisplaySummary[]>(`/summaries/${pageFileName(i)}`), + enabled: pageCount > 0, + })), }); - return promise; + + const loadedPages = pageQueries.filter((q) => q.data).length; + const summariesReady = pageCount > 0 && loadedPages === pageCount; + + // Concatenate available pages; sort once all pages are in. Pages may arrive + // out of order, so a late sort keeps the slug-descending (newest-first) + // ordering stable regardless of arrival order. + const summaries = useMemo<DisplaySummary[]>(() => { + if (loadedPages === 0) return []; + const out: DisplaySummary[] = []; + for (const q of pageQueries) if (q.data) out.push(...q.data); + if (summariesReady) out.sort((a, b) => b.slug.localeCompare(a.slug)); + return out; + // pageQueries identity changes every render; key off loadedPages + ready. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [loadedPages, summariesReady]); + + const error = + manifestQuery.error instanceof Error + ? manifestQuery.error + : (pageQueries.find((q) => q.error)?.error as Error | undefined) ?? null; + + return { + manifest, + summaries, + loadedPages, + pageCount, + summariesReady, + error, + }; } diff --git a/app/transcriptCache.ts b/app/transcriptCache.ts @@ -2,20 +2,38 @@ import type { TranscriptDetail } from "@/lib/transcripts"; +// Cap chosen to cover realistic "recently viewed" sessions (~3 MB at ~30 KB +// per transcript) without pinning the full corpus during a search scan. +const MAX_ENTRIES = 100; + +// Map's insertion order gives us a free FIFO; we promote on access +// (delete + set) to approximate LRU so the most recently used entries stay. const cache = new Map<string, Promise<TranscriptDetail>>(); +function promote(slug: string, p: Promise<TranscriptDetail>): void { + cache.delete(slug); + cache.set(slug, p); + while (cache.size > MAX_ENTRIES) { + const oldest = cache.keys().next().value; + if (oldest === undefined) break; + cache.delete(oldest); + } +} + export function fetchTranscript(slug: string): Promise<TranscriptDetail> { - const hit = cache.get(slug); - if (hit) return hit; - const p = fetch(`/transcripts/${slug}`).then((r) => { + const existing = cache.get(slug); + if (existing) { + promote(slug, existing); + return existing; + } + const p = fetch(`/transcripts/${slug}.json`).then((r) => { if (!r.ok) throw new Error(`Failed to fetch transcript ${slug}`); return r.json() as Promise<TranscriptDetail>; }); - cache.set(slug, p); - p.catch(() => cache.delete(slug)); + // Evict on failure so retries can go over the wire again. + p.catch(() => { + if (cache.get(slug) === p) cache.delete(slug); + }); + promote(slug, p); return p; } - -export function peekTranscript(slug: string): Promise<TranscriptDetail> | undefined { - return cache.get(slug); -} diff --git a/app/transcripts/[slug]/route.ts b/app/transcripts/[slug]/route.ts @@ -1,15 +0,0 @@ -import { listSlugs, loadTranscript } from "@/lib/transcripts"; - -export const dynamic = "force-static"; - -export async function generateStaticParams() { - return (await listSlugs()).map((slug) => ({ slug })); -} - -export async function GET( - _req: Request, - { params }: { params: Promise<{ slug: string }> }, -) { - const { slug } = await params; - return Response.json(await loadTranscript(slug)); -} diff --git a/app/urlState.ts b/app/urlState.ts @@ -1,12 +1,17 @@ "use client"; -import { useSyncExternalStore } from "react"; +import { useMemo, useSyncExternalStore } from "react"; export type UrlParams = { q: string; re: boolean; v: string | null; t: number | null; + ch: string[]; + nov: boolean; + nol: boolean; + naa: boolean; + nar: boolean; }; const listeners = new Set<() => void>(); @@ -37,12 +42,17 @@ function parse(search: string): UrlParams { re: p.get("re") === "1", v: p.get("v"), t: t !== null && Number.isFinite(t) ? t : null, + ch: p.getAll("ch"), + nov: p.get("nov") === "1", + nol: p.get("nol") === "1", + naa: p.get("naa") === "1", + nar: p.get("nar") === "1", }; } export function useUrlParams(): UrlParams { const search = useSyncExternalStore(subscribe, getSnapshot, getServerSnapshot); - return parse(search); + return useMemo(() => parse(search), [search]); } type Patch = Partial<{ @@ -50,6 +60,11 @@ type Patch = Partial<{ re: boolean; v: string | null; t: number | null; + ch: string[]; + nov: boolean; + nol: boolean; + naa: boolean; + nar: boolean; }>; export function writeUrlParams(patch: Patch) { @@ -70,6 +85,26 @@ export function writeUrlParams(patch: Patch) { if (patch.t !== null) params.set("t", String(patch.t)); else params.delete("t"); } + if (patch.ch !== undefined) { + params.delete("ch"); + for (const v of patch.ch) params.append("ch", v); + } + if (patch.nov !== undefined) { + if (patch.nov) params.set("nov", "1"); + else params.delete("nov"); + } + if (patch.nol !== undefined) { + if (patch.nol) params.set("nol", "1"); + else params.delete("nol"); + } + if (patch.naa !== undefined) { + if (patch.naa) params.set("naa", "1"); + else params.delete("naa"); + } + if (patch.nar !== undefined) { + if (patch.nar) params.set("nar", "1"); + else params.delete("nar"); + } const qs = params.toString(); const next = `${window.location.pathname}${qs ? `?${qs}` : ""}`; if (next === window.location.pathname + window.location.search) return; diff --git a/lib/manifest.ts b/lib/manifest.ts @@ -0,0 +1,20 @@ +export type ChannelEntry = { + name: string; + count: number; +}; + +export type Manifest = { + version: number; + totalCount: number; + pageSize: number; + pageCount: number; + generatedAt: string; + channels: ChannelEntry[]; +}; + +export const MANIFEST_VERSION = 1; +export const SUMMARIES_PAGE_SIZE = 1000; + +export function pageFileName(index: number): string { + return `page-${String(index).padStart(4, "0")}.json`; +} diff --git a/lib/transcripts-server.ts b/lib/transcripts-server.ts @@ -0,0 +1,52 @@ +import { formatDate, formatDuration } from "./format"; +import type { DisplaySummary, TranscriptSummary } from "./transcripts"; + +export type RawMetadata = { + id?: string; + title?: string; + upload_date?: string; + duration?: number; + channel?: string; + uploader?: string; + description?: string; + is_live?: boolean; + was_live?: boolean; + live_status?: string; + age_limit?: number; +}; + +export function summarize(slug: string, meta: RawMetadata): TranscriptSummary { + const dateFromSlug = slug.match(/^(\d{8})_/)?.[1]; + const liveStatus = meta.live_status ?? ""; + const isLivestream = + meta.was_live === true || + meta.is_live === true || + liveStatus === "was_live" || + liveStatus === "is_live" || + liveStatus === "is_upcoming"; + return { + slug, + id: meta.id ?? slug.split("_")[1]?.split("-")[0] ?? slug, + title: meta.title ?? slug, + uploadDate: meta.upload_date ?? dateFromSlug ?? "", + duration: meta.duration ?? 0, + channel: meta.channel ?? meta.uploader ?? "", + description: meta.description ?? "", + isLivestream, + ageRestricted: (meta.age_limit ?? 0) > 0, + }; +} + +export function toDisplaySummary(t: TranscriptSummary): DisplaySummary { + return { + slug: t.slug, + id: t.id, + title: t.title, + uploadDate: t.uploadDate, + date: formatDate(t.uploadDate), + duration: formatDuration(t.duration), + channel: t.channel, + isLivestream: t.isLivestream, + ageRestricted: t.ageRestricted, + }; +} diff --git a/lib/transcripts.ts b/lib/transcripts.ts @@ -1,9 +1,13 @@ import path from "node:path"; -import { parseVtt, type Cue } from "./vtt"; -import { readFile, readdir } from "fs-extra"; -import { formatDate, formatDuration } from "./format"; +import type { Cue } from "./vtt"; +import { readFile } from "fs-extra"; -const ROOT = path.join(process.cwd(), "transcripts", "data"); +const MANIFEST_PATH = path.join( + process.cwd(), + "public", + "summaries", + "manifest.json", +); export type TranscriptSummary = { slug: string; @@ -13,6 +17,8 @@ export type TranscriptSummary = { duration: number; channel: string; description: string; + isLivestream: boolean; + ageRestricted: boolean; }; export type DisplaySummary = { @@ -23,88 +29,21 @@ export type DisplaySummary = { date: string; duration: string; channel: string; + isLivestream: boolean; + ageRestricted: boolean; }; export type TranscriptDetail = TranscriptSummary & { cues: Cue[] | undefined; }; -type RawMetadata = { - id?: string; - title?: string; - upload_date?: string; - duration?: number; - channel?: string; - uploader?: string; - description?: string; -}; - -async function readMetadata(slug: string): Promise<RawMetadata> { - const file = path.join(ROOT, slug, "metadata.info.json"); - const raw = await readFile(file, "utf8"); - return JSON.parse(raw) as RawMetadata; -} - -async function readTranscript(slug: string): Promise<Cue[] | undefined> { - const dir = path.join(ROOT, slug); - const vttPath = path.join(dir, "transcript.en.vtt"); - try { - const fileContents = await readFile(vttPath, "utf8"); - return parseVtt(fileContents); - } catch { - return undefined; - } -} - -function summarize(slug: string, meta: RawMetadata): TranscriptSummary { - const dateFromSlug = slug.match(/^(\d{8})_/)?.[1]; - return { - slug, - id: meta.id ?? slug.split("_")[1]?.split("-")[0] ?? slug, - title: meta.title ?? slug, - uploadDate: meta.upload_date ?? dateFromSlug ?? "", - duration: meta.duration ?? 0, - channel: meta.channel ?? meta.uploader ?? "", - description: meta.description ?? "", - }; -} - -export async function listSlugs(): Promise<string[]> { - return (await readdir(ROOT, { withFileTypes: true })) - .filter((d) => d.isDirectory()) - .map((d) => d.name) - .sort((a, b) => b.localeCompare(a)); -} - -export async function listTranscripts(): Promise<TranscriptSummary[]> { - return await Promise.all( - (await listSlugs()).map(async (slug) => - summarize(slug, await readMetadata(slug)), - ), - ); -} - export async function countTranscripts(): Promise<number> { - return (await listSlugs()).length; -} - -export async function listDisplaySummaries(): Promise<DisplaySummary[]> { - const summaries = await listTranscripts(); - return summaries.map((t) => ({ - slug: t.slug, - id: t.id, - title: t.title, - uploadDate: t.uploadDate, - date: formatDate(t.uploadDate), - duration: formatDuration(t.duration), - channel: t.channel, - })); -} - -export async function loadTranscript(slug: string): Promise<TranscriptDetail> { - const meta = await readMetadata(slug); - const cues = await readTranscript(slug); - return { ...summarize(slug, meta), cues }; + const raw = await readFile(MANIFEST_PATH, "utf8"); + const parsed = JSON.parse(raw) as { totalCount?: number }; + if (typeof parsed.totalCount !== "number") { + throw new Error("manifest.json missing totalCount; run `pnpm build:index`"); + } + return parsed.totalCount; } export { formatDate, formatDuration } from "./format"; diff --git a/next.config.ts b/next.config.ts @@ -4,6 +4,7 @@ const nextConfig: NextConfig = { output: "export", trailingSlash: true, images: { unoptimized: true }, + staticPageGenerationTimeout: 600, // Increase timeout to 600 seconds }; export default nextConfig; diff --git a/package.json b/package.json @@ -4,11 +4,15 @@ "private": true, "scripts": { "dev": "next dev", + "build:index": "tsx scripts/build-index.ts", + "prebuild": "pnpm build:index", "build": "next build", "start": "serve out", "lint": "eslint" }, "dependencies": { + "@tanstack/react-query": "^5.99.0", + "lmdb": "^3.1.5", "next": "16.2.3", "react": "19.2.4", "react-dom": "19.2.4", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -8,6 +8,12 @@ importers: .: dependencies: + '@tanstack/react-query': + specifier: ^5.99.0 + version: 5.99.0(react@19.2.4) + lmdb: + specifier: ^3.1.5 + version: 3.5.4 next: specifier: 16.2.3 version: 16.2.3(@babel/core@7.29.0)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) @@ -334,6 +340,9 @@ packages: resolution: {integrity: sha512-43/qtrDUokr7LJqoF2c3+RInu/t4zfrpYdoSDfYyhg52rwLV6TnOvdG4fXm7IkSB3wErkcmJS9iEhjVtOSEjjA==} engines: {node: ^18.18.0 || ^20.9.0 || >=21.1.0} + '@harperfast/extended-iterable@1.0.3': + resolution: {integrity: sha512-sSAYhQca3rDWtQUHSAPeO7axFIUJOI6hn1gjRC5APVE1a90tuyT8f5WIgRsFhhWA7htNkju2veB9eWL6YHi/Lw==} + '@humanfs/core@0.19.1': resolution: {integrity: sha512-5DyQ4+1JEUzejeK1JGICcideyfUbGixgS9jNgex5nqkW+cY7WZhxBigmieN5Qnw9ZosSNVC9KQKyb+GUaGyKUA==} engines: {node: '>=18.18.0'} @@ -519,6 +528,71 @@ packages: '@jridgewell/trace-mapping@0.3.31': resolution: {integrity: sha512-zzNR+SdQSDJzc8joaeP8QQoCQr8NuYx2dIIytl1QeBEZHJ9uW6hebsrYgbz8hJwUQao3TWCMtmfV8Nu1twOLAw==} + '@lmdb/lmdb-darwin-arm64@3.5.4': + resolution: {integrity: sha512-Kk4Kz3iyu1QiLsLZBS9Af1eSKUC8VR2T+/jyE2iAyuGw2VwK08pp5iTbZnXn6sWu0LogO/RFktMxOjiDA2sS3w==} + cpu: [arm64] + os: [darwin] + + '@lmdb/lmdb-darwin-x64@3.5.4': + resolution: {integrity: sha512-BEe5Rp3trn26oxoXOVL5HVDoiYmjUDwr8NRPkBOdUdCSBEorKI+7JrZLRKAdxO+G6cGQLgseXk0gR7qIQa7aGw==} + cpu: [x64] + os: [darwin] + + '@lmdb/lmdb-linux-arm64@3.5.4': + resolution: {integrity: sha512-cUXEengO8o60v1SWerJTH4/RH4U3+9jC0/4njp2Z9NdmvaGzhKsbRM2wpXuRYrN8tytsoJCg0SvWEWwHAwLbCA==} + cpu: [arm64] + os: [linux] + + '@lmdb/lmdb-linux-arm@3.5.4': + resolution: {integrity: sha512-SGbFR7816uBcTHc2ZY4S6WyOkl9bICnzqTQd2Mv4V/j24cfds88xx2nC6cm/y8zGQL7Ds31YF/5NGxjgcdM5Hw==} + cpu: [arm] + os: [linux] + + '@lmdb/lmdb-linux-x64@3.5.4': + resolution: {integrity: sha512-Gxq8jpgOWXwd0PUR+c9R2Ik1/uBnGd5GMIIzRRDqABCkvmjtC3KWcyhesV9jSPCz759isl0NlbsstZ2oyvk8lA==} + cpu: [x64] + os: [linux] + + '@lmdb/lmdb-win32-arm64@3.5.4': + resolution: {integrity: sha512-pKv1DJ1bPZAaHkdFsSz5IDfUG8x9vntgquXF9/Dm2xuupcIe/EkLzylpoBxppFVK5vzbV561Dq26jNY2fIMA7g==} + cpu: [arm64] + os: [win32] + + '@lmdb/lmdb-win32-x64@3.5.4': + resolution: {integrity: sha512-JF1BmLCm9kGEVZgYmJq43zeQVdHVgAJnTi/NURWEsy6L1ZrrlSmdltS+D17QN4LODwf+1LMXAA9auIZVXtWwzw==} + cpu: [x64] + os: [win32] + + '@msgpackr-extract/msgpackr-extract-darwin-arm64@3.0.3': + resolution: {integrity: sha512-QZHtlVgbAdy2zAqNA9Gu1UpIuI8Xvsd1v8ic6B2pZmeFnFcMWiPLfWXh7TVw4eGEZ/C9TH281KwhVoeQUKbyjw==} + cpu: [arm64] + os: [darwin] + + '@msgpackr-extract/msgpackr-extract-darwin-x64@3.0.3': + resolution: {integrity: sha512-mdzd3AVzYKuUmiWOQ8GNhl64/IoFGol569zNRdkLReh6LRLHOXxU4U8eq0JwaD8iFHdVGqSy4IjFL4reoWCDFw==} + cpu: [x64] + os: [darwin] + + '@msgpackr-extract/msgpackr-extract-linux-arm64@3.0.3': + resolution: {integrity: sha512-YxQL+ax0XqBJDZiKimS2XQaf+2wDGVa1enVRGzEvLLVFeqa5kx2bWbtcSXgsxjQB7nRqqIGFIcLteF/sHeVtQg==} + cpu: [arm64] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-linux-arm@3.0.3': + resolution: {integrity: sha512-fg0uy/dG/nZEXfYilKoRe7yALaNmHoYeIoJuJ7KJ+YyU2bvY8vPv27f7UKhGRpY6euFYqEVhxCFZgAUNQBM3nw==} + cpu: [arm] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-linux-x64@3.0.3': + resolution: {integrity: sha512-cvwNfbP07pKUfq1uH+S6KJ7dT9K8WOE4ZiAcsrSes+UY55E/0jLYc+vq+DO7jlmqRb5zAggExKm0H7O/CBaesg==} + cpu: [x64] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-win32-x64@3.0.3': + resolution: {integrity: sha512-x0fWaQtYp4E6sktbsdAqnehxDgEc/VwM7uLsRCYWaiGu0ykYdZPiS8zCWdnjHwyiumousxfBm4SO31eXqwEZhQ==} + cpu: [x64] + os: [win32] + '@napi-rs/wasm-runtime@0.2.12': resolution: {integrity: sha512-ZVWUcfwY4E/yPitQJl481FjFo3K22D6qF0DuFH6Y/nbnE11GY5uguDxZMGXPQ8WQ0128MXQD7TnfHyK4oWoIJQ==} @@ -694,6 +768,14 @@ packages: '@tailwindcss/postcss@4.2.2': resolution: {integrity: sha512-n4goKQbW8RVXIbNKRB/45LzyUqN451deQK0nzIeauVEqjlI49slUlgKYJM2QyUzap/PcpnS7kzSUmPb1sCRvYQ==} + '@tanstack/query-core@5.99.0': + resolution: {integrity: sha512-3Jv3WQG0BCcH7G+7lf/bP8QyBfJOXeY+T08Rin3GZ1bshvwlbPt7NrDHMEzGdKIOmOzvIQmxjk28YEQX60k7pQ==} + + '@tanstack/react-query@5.99.0': + resolution: {integrity: sha512-OY2bCqPemT1LlqJ8Y2CUau4KELnIhhG9Ol3ZndPbdnB095pRbPo1cHuXTndg8iIwtoHTgwZjyaDnQ0xD0mYwAw==} + peerDependencies: + react: ^18 || ^19 + '@tybys/wasm-util@0.10.1': resolution: {integrity: sha512-9tTaPJLSiejZKx+Bmog4uSubteqTvFrVrURwkmHixBo0G4seD0zUxp98E1DzUBJxLQ3NPwXrGKDiVjwx/DpPsg==} @@ -1811,6 +1893,10 @@ packages: resolution: {integrity: sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ==} engines: {node: '>= 12.0.0'} + lmdb@3.5.4: + resolution: {integrity: sha512-9FKQA6G1MMtqNxfxvSBNXD/axeG2QRjYbNh0/ykRL5xYcRbCm2vXq7B9bhc7nSuKdHzr8/BHIwfPuYYH1UsXXw==} + hasBin: true + load-script@1.0.0: resolution: {integrity: sha512-kPEjMFtZvwL9TaZo0uZ2ml+Ye9HUMmPwbYRJ324qF9tqMejwykJ5ggTyvzmrbBeapCAbk98BSbTeovHEEP1uCA==} @@ -1881,6 +1967,13 @@ packages: ms@2.1.3: resolution: {integrity: sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==} + msgpackr-extract@3.0.3: + resolution: {integrity: sha512-P0efT1C9jIdVRefqjzOQ9Xml57zpOXnIuS+csaB4MdZbTdmGDLo8XhzBG1N7aO11gKDDkJvBLULeFTo46wwreA==} + hasBin: true + + msgpackr@1.11.9: + resolution: {integrity: sha512-FkoAAyyA6HM8wL882EcEyFZ9s7hVADSwG9xrVx3dxxNQAtgADTrJoEWivID82Iv1zWDsv/OtbrrcZAzGzOMdNw==} + nanoid@3.3.11: resolution: {integrity: sha512-N8SpfPUnUp1bK+PMYW8qSWdl9U+wwNWI4QKxOYDy9JAro3WMX7p2OeVRF9v+347pnakNevPmiHhNmZ2HbFA76w==} engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} @@ -1919,10 +2012,17 @@ packages: sass: optional: true + node-addon-api@6.1.0: + resolution: {integrity: sha512-+eawOlIgy680F0kBzPUNFhMZGtJ1YmqM6l4+Crf4IkImjYrO/mqPwRMh352g23uIaQKFItcQ64I7KMaJxHgAVA==} + node-exports-info@1.6.0: resolution: {integrity: sha512-pyFS63ptit/P5WqUkt+UUfe+4oevH+bFeIiPPdfb0pFeYEu/1ELnJu5l+5EcTKYL5M7zaAa7S8ddywgXypqKCw==} engines: {node: '>= 0.4'} + node-gyp-build-optional-packages@5.2.2: + resolution: {integrity: sha512-s+w+rBWnpTMwSFbaE0UXsRlg7hU4FjekKU4eyAih5T8nJuNZT1nNsskXpxmeqSK9UzkBl6UgRlnKc8hz8IEqOw==} + hasBin: true + node-releases@2.0.37: resolution: {integrity: sha512-1h5gKZCF+pO/o3Iqt5Jp7wc9rH3eJJ0+nh/CIoiRwjRxde/hAHyLPXYN4V3CqKAbiZPSeJFSWHmJsbkicta0Eg==} @@ -1974,6 +2074,9 @@ packages: resolution: {integrity: sha512-6IpQ7mKUxRcZNLIObR0hz7lxsapSSIYNZJwXPGeF0mTVqGKFIXj1DQcMoT22S3ROcLyY/rz0PWaWZ9ayWmad9g==} engines: {node: '>= 0.8.0'} + ordered-binary@1.6.1: + resolution: {integrity: sha512-QkCdPooczexPLiXIrbVOPYkR3VO3T6v2OyKRkR1Xbhpy7/LAVXwahnRCgRp78Oe/Ehf0C/HATAxfSr6eA1oX+w==} + own-keys@1.0.1: resolution: {integrity: sha512-qFOyK5PjiWZd+QQIh+1jhdb9LpxTF0qs7Pm8o5QHYZ0M3vKqSqzsZaEB6oWlxZ+q2sJBMI/Ktgd2N5ZwQoRHfg==} engines: {node: '>= 0.4'} @@ -2373,6 +2476,9 @@ packages: resolution: {integrity: sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg==} engines: {node: '>= 0.8'} + weak-lru-cache@1.2.2: + resolution: {integrity: sha512-DEAoo25RfSYMuTGc9vPJzZcZullwIqRDSI9LOy+fkCJPi6hykCnfKaXTuPBDuXAUcqHXyOgFtHNp/kB2FjYHbw==} + which-boxed-primitive@1.1.1: resolution: {integrity: sha512-TbX3mj8n0odCBFVlY8AxkqcHASw3L60jIuF8jFP78az3C2YhmGvqbHBpAjTRH2/xqYunrJ9g1jSyjCjpoWzIAA==} engines: {node: '>= 0.4'} @@ -2666,6 +2772,8 @@ snapshots: '@eslint/core': 0.17.0 levn: 0.4.1 + '@harperfast/extended-iterable@1.0.3': {} + '@humanfs/core@0.19.1': {} '@humanfs/node@0.16.7': @@ -2793,6 +2901,45 @@ snapshots: '@jridgewell/resolve-uri': 3.1.2 '@jridgewell/sourcemap-codec': 1.5.5 + '@lmdb/lmdb-darwin-arm64@3.5.4': + optional: true + + '@lmdb/lmdb-darwin-x64@3.5.4': + optional: true + + '@lmdb/lmdb-linux-arm64@3.5.4': + optional: true + + '@lmdb/lmdb-linux-arm@3.5.4': + optional: true + + '@lmdb/lmdb-linux-x64@3.5.4': + optional: true + + '@lmdb/lmdb-win32-arm64@3.5.4': + optional: true + + '@lmdb/lmdb-win32-x64@3.5.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-darwin-arm64@3.0.3': + optional: true + + '@msgpackr-extract/msgpackr-extract-darwin-x64@3.0.3': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-arm64@3.0.3': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-arm@3.0.3': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-x64@3.0.3': + optional: true + + '@msgpackr-extract/msgpackr-extract-win32-x64@3.0.3': + optional: true + '@napi-rs/wasm-runtime@0.2.12': dependencies: '@emnapi/core': 1.9.2 @@ -2919,6 +3066,13 @@ snapshots: postcss: 8.5.9 tailwindcss: 4.2.2 + '@tanstack/query-core@5.99.0': {} + + '@tanstack/react-query@5.99.0(react@19.2.4)': + dependencies: + '@tanstack/query-core': 5.99.0 + react: 19.2.4 + '@tybys/wasm-util@0.10.1': dependencies: tslib: 2.8.1 @@ -4184,6 +4338,23 @@ snapshots: lightningcss-win32-arm64-msvc: 1.32.0 lightningcss-win32-x64-msvc: 1.32.0 + lmdb@3.5.4: + dependencies: + '@harperfast/extended-iterable': 1.0.3 + msgpackr: 1.11.9 + node-addon-api: 6.1.0 + node-gyp-build-optional-packages: 5.2.2 + ordered-binary: 1.6.1 + weak-lru-cache: 1.2.2 + optionalDependencies: + '@lmdb/lmdb-darwin-arm64': 3.5.4 + '@lmdb/lmdb-darwin-x64': 3.5.4 + '@lmdb/lmdb-linux-arm': 3.5.4 + '@lmdb/lmdb-linux-arm64': 3.5.4 + '@lmdb/lmdb-linux-x64': 3.5.4 + '@lmdb/lmdb-win32-arm64': 3.5.4 + '@lmdb/lmdb-win32-x64': 3.5.4 + load-script@1.0.0: {} locate-path@6.0.0: @@ -4241,6 +4412,22 @@ snapshots: ms@2.1.3: {} + msgpackr-extract@3.0.3: + dependencies: + node-gyp-build-optional-packages: 5.2.2 + optionalDependencies: + '@msgpackr-extract/msgpackr-extract-darwin-arm64': 3.0.3 + '@msgpackr-extract/msgpackr-extract-darwin-x64': 3.0.3 + '@msgpackr-extract/msgpackr-extract-linux-arm': 3.0.3 + '@msgpackr-extract/msgpackr-extract-linux-arm64': 3.0.3 + '@msgpackr-extract/msgpackr-extract-linux-x64': 3.0.3 + '@msgpackr-extract/msgpackr-extract-win32-x64': 3.0.3 + optional: true + + msgpackr@1.11.9: + optionalDependencies: + msgpackr-extract: 3.0.3 + nanoid@3.3.11: {} napi-postinstall@0.3.4: {} @@ -4273,6 +4460,8 @@ snapshots: - '@babel/core' - babel-plugin-macros + node-addon-api@6.1.0: {} + node-exports-info@1.6.0: dependencies: array.prototype.flatmap: 1.3.3 @@ -4280,6 +4469,10 @@ snapshots: object.entries: 1.1.9 semver: 6.3.1 + node-gyp-build-optional-packages@5.2.2: + dependencies: + detect-libc: 2.1.2 + node-releases@2.0.37: {} npm-run-path@4.0.1: @@ -4343,6 +4536,8 @@ snapshots: type-check: 0.4.0 word-wrap: 1.2.5 + ordered-binary@1.6.1: {} + own-keys@1.0.1: dependencies: get-intrinsic: 1.3.0 @@ -4862,6 +5057,8 @@ snapshots: vary@1.1.2: {} + weak-lru-cache@1.2.2: {} + which-boxed-primitive@1.1.1: dependencies: is-bigint: 1.1.0 diff --git a/scripts/build-index.ts b/scripts/build-index.ts @@ -0,0 +1,299 @@ +#!/usr/bin/env tsx +// Preprocess transcripts/data/* into: +// - LMDB cache at transcripts/index.mdb (for incremental rebuilds) +// - public/summaries/{manifest,page-NNNN}.json (paginated summaries index) +// - public/transcripts/<slug>.json (per-transcript cues + metadata) +// +// Short-circuits when mtimes already match and the public outputs are intact. + +import path from "node:path"; +import { mkdir, readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises"; +import { open } from "lmdb"; +import { parseVtt, type Cue } from "../lib/vtt"; +import { summarize, toDisplaySummary, type RawMetadata } from "../lib/transcripts-server"; +import type { TranscriptSummary, DisplaySummary } from "../lib/transcripts"; +import type { Manifest, ChannelEntry } from "../lib/manifest"; +import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE, pageFileName } from "../lib/manifest"; + +const ROOT = process.cwd(); +const DATA_DIR = path.join(ROOT, "transcripts", "data"); +const DB_PATH = path.join(ROOT, "transcripts", "index.mdb"); +const PUBLIC_DIR = path.join(ROOT, "public"); +const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries"); +const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts"); +const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json"); + +const SCHEMA_VERSION = 1; + +type MtimeRecord = { metaMs: number; vttMs: number | null }; + +type LiveSlug = { + slug: string; + metaPath: string; + metaMs: number; + vttPath: string; + vttMs: number | null; +}; + +async function exists(p: string): Promise<boolean> { + try { + await stat(p); + return true; + } catch { + return false; + } +} + +async function scanSource(): Promise<LiveSlug[]> { + const entries = await readdir(DATA_DIR, { withFileTypes: true }); + const out: LiveSlug[] = []; + for (const e of entries) { + if (!e.isDirectory()) continue; + const slug = e.name; + const metaPath = path.join(DATA_DIR, slug, "metadata.info.json"); + const vttPath = path.join(DATA_DIR, slug, "transcript.en.vtt"); + let metaMs: number; + try { + metaMs = (await stat(metaPath)).mtimeMs; + } catch { + // Skip slugs without metadata — they're not usable. + continue; + } + let vttMs: number | null = null; + try { + vttMs = (await stat(vttPath)).mtimeMs; + } catch { + vttMs = null; + } + out.push({ slug, metaPath, metaMs, vttPath, vttMs }); + } + return out; +} + +async function writeJsonAtomic(filePath: string, value: unknown): Promise<void> { + const tmp = `${filePath}.tmp-${process.pid}`; + await writeFile(tmp, JSON.stringify(value)); + await rename(tmp, filePath); +} + +async function main(): Promise<void> { + const t0 = Date.now(); + await mkdir(path.dirname(DB_PATH), { recursive: true }); + await mkdir(SUMMARIES_DIR, { recursive: true }); + await mkdir(TRANSCRIPTS_DIR, { recursive: true }); + + const root = open({ + path: DB_PATH, + maxDbs: 8, + compression: true, + }); + const sums = root.openDB<TranscriptSummary, string>({ + name: "sums", + encoding: "msgpack", + }); + const cues = root.openDB<Cue[], string>({ + name: "cues", + encoding: "msgpack", + }); + const mtimes = root.openDB<MtimeRecord, string>({ + name: "mtimes", + encoding: "msgpack", + }); + const meta = root.openDB<unknown, string>({ + name: "meta", + encoding: "msgpack", + }); + + // Force full rebuild if the schema changed. + const storedSchema = meta.get("schema") as number | undefined; + const schemaBumped = storedSchema !== SCHEMA_VERSION; + if (schemaBumped) { + console.log( + `Schema change (${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}); invalidating LMDB cache.`, + ); + await sums.clearAsync(); + await cues.clearAsync(); + await mtimes.clearAsync(); + await meta.put("schema", SCHEMA_VERSION); + } + + const live = await scanSource(); + const liveBySlug = new Map(live.map((s) => [s.slug, s])); + + const known = new Set<string>(); + for (const { key } of mtimes.getRange()) known.add(key); + + const added: LiveSlug[] = []; + const changed: LiveSlug[] = []; + const removed: string[] = []; + + for (const s of live) { + const prev = mtimes.get(s.slug); + if (!prev) { + added.push(s); + } else if (prev.metaMs !== s.metaMs || prev.vttMs !== s.vttMs) { + changed.push(s); + } + } + for (const slug of known) { + if (!liveBySlug.has(slug)) removed.push(slug); + } + + const anyMutations = + added.length > 0 || changed.length > 0 || removed.length > 0; + + // Short-circuit: if nothing changed AND public/ is intact, we're done. + if (!anyMutations && !schemaBumped) { + const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null); + if (manifestRaw) { + try { + const parsed = JSON.parse(manifestRaw) as Manifest; + if ( + parsed.version === MANIFEST_VERSION && + parsed.totalCount === live.length && + parsed.pageSize === SUMMARIES_PAGE_SIZE + ) { + // Spot-check first and last page files exist. + const firstPage = path.join(SUMMARIES_DIR, pageFileName(0)); + const lastPage = path.join( + SUMMARIES_DIR, + pageFileName(Math.max(0, parsed.pageCount - 1)), + ); + if ((await exists(firstPage)) && (await exists(lastPage))) { + console.log( + `Index up to date (${live.length} transcripts, ${parsed.pageCount} pages). Skipping.`, + ); + await root.close(); + return; + } + } + } catch { + // fall through to full emission + } + } + } + + console.log( + `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, + ); + + // Process mutations: parse files and write to LMDB + per-slug public JSON. + const toProcess = [...added, ...changed]; + let processed = 0; + const BATCH = 200; + for (let i = 0; i < toProcess.length; i += BATCH) { + const slice = toProcess.slice(i, i + BATCH); + await Promise.all( + slice.map(async (s) => { + try { + const metaRaw = await readFile(s.metaPath, "utf8"); + const parsedMeta = JSON.parse(metaRaw) as RawMetadata; + const summary = summarize(s.slug, parsedMeta); + let cueList: Cue[] | undefined; + if (s.vttMs !== null) { + try { + const vtt = await readFile(s.vttPath, "utf8"); + cueList = parseVtt(vtt); + } catch { + cueList = undefined; + } + } + sums.put(s.slug, summary); + if (cueList) cues.put(s.slug, cueList); + else cues.remove(s.slug); + mtimes.put(s.slug, { metaMs: s.metaMs, vttMs: s.vttMs }); + + const detail = { ...summary, cues: cueList }; + await writeJsonAtomic( + path.join(TRANSCRIPTS_DIR, `${s.slug}.json`), + detail, + ); + } catch (err) { + console.warn(`Failed to process ${s.slug}:`, err); + } + }), + ); + processed += slice.length; + if (toProcess.length > BATCH) { + process.stdout.write(` processed ${processed}/${toProcess.length}\r`); + } + } + if (toProcess.length > BATCH) process.stdout.write("\n"); + + // Process removals. + for (const slug of removed) { + sums.remove(slug); + cues.remove(slug); + mtimes.remove(slug); + await rm(path.join(TRANSCRIPTS_DIR, `${slug}.json`), { force: true }); + } + + // Flush writes before we iterate for emission. + await sums.flushed; + await cues.flushed; + await mtimes.flushed; + + // Stream LMDB (reverse slug order = newest first) to emit paginated summaries. + // Hold at most one page worth of summaries in memory at once. + const pageSize = SUMMARIES_PAGE_SIZE; + const channels = new Map<string, number>(); + let pageIndex = 0; + let buffer: DisplaySummary[] = []; + let total = 0; + + const flushPage = async () => { + if (buffer.length === 0) return; + const p = path.join(SUMMARIES_DIR, pageFileName(pageIndex)); + await writeJsonAtomic(p, buffer); + buffer = []; + pageIndex++; + }; + + // LMDB returns keys in ascending order; we want newest-first (descending slug). + // Slugs are `YYYYMMDD_...` so lexicographic reverse == chronological reverse. + for (const { value } of sums.getRange({ reverse: true })) { + const s = value as TranscriptSummary; + if (s.channel) channels.set(s.channel, (channels.get(s.channel) ?? 0) + 1); + buffer.push(toDisplaySummary(s)); + total++; + if (buffer.length >= pageSize) await flushPage(); + } + await flushPage(); + + // Prune stale page files (if page count shrank). + const expectedPages = new Set<string>(); + for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i)); + const existing = await readdir(SUMMARIES_DIR).catch(() => [] as string[]); + for (const name of existing) { + if (!name.startsWith("page-")) continue; + if (!expectedPages.has(name)) { + await rm(path.join(SUMMARIES_DIR, name), { force: true }); + } + } + + const channelList: ChannelEntry[] = Array.from(channels.entries()) + .map(([name, count]) => ({ name, count })) + .sort((a, b) => a.name.localeCompare(b.name)); + + const manifest: Manifest = { + version: MANIFEST_VERSION, + totalCount: total, + pageSize, + pageCount: pageIndex, + generatedAt: new Date().toISOString(), + channels: channelList, + }; + await writeJsonAtomic(MANIFEST_PATH, manifest); + + await root.close(); + + const secs = ((Date.now() - t0) / 1000).toFixed(2); + console.log( + `Done in ${secs}s. ${total} transcripts across ${pageIndex} pages; ${channelList.length} channels.`, + ); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/serve.json b/serve.json @@ -0,0 +1,14 @@ +{ + "headers": [ + { + "source": "**/*.json", + "headers": [ + { + "key": "Cache-Control", + "value": "public, max-age=3600" + } + ] + } + ] +} +