Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 552a70c17bb05455caa3e21be7dd7e89eff79343
parent 19defd08da573ceab111e94fa67f4a32e029cb05
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 21 Apr 2026 07:38:40 -0400

Add Rumble support

Diffstat:
Mapp/PlayerProvider.tsx | 108++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------
Aapp/RumblePlayer.tsx | 47+++++++++++++++++++++++++++++++++++++++++++++++
Mapp/TranscriptModal.tsx | 31+++++++++++++++++++++++++++++++
Mapp/summariesCache.ts | 14+++++++++++---
Mlib/transcripts-server.ts | 43+++++++++++++++++++++++++++++++++++--------
Mlib/transcripts.ts | 8++++++++
Alib/whisper.ts | 32++++++++++++++++++++++++++++++++
Mpackage.json | 1+
Mpnpm-lock.yaml | 24++++++++++++++++++++++++
Mscripts/build-index.ts | 295+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------
10 files changed, 499 insertions(+), 104 deletions(-)

diff --git a/app/PlayerProvider.tsx b/app/PlayerProvider.tsx @@ -14,11 +14,19 @@ import type ReactPlayerType from "react-player"; import { fetchTranscript } from "./transcriptCache"; import { useUrlParams, writeUrlParams } from "./urlState"; import { formatDate, formatDuration } from "@/lib/format"; +import type { Platform } from "@/lib/transcripts"; +import type { RumblePlayerHandle } from "./RumblePlayer"; const ReactPlayer = dynamic(() => import("react-player/youtube"), { ssr: false, }); +const RumblePlayer = dynamic(() => import("./RumblePlayer"), { + ssr: false, +}); + +type PlayerHandle = Pick<ReactPlayerType, "seekTo"> | RumblePlayerHandle; + export type Cue = { start: number; end: number; text: string }; export type TranscriptData = { @@ -32,6 +40,8 @@ export type TranscriptData = { description?: string; isLivestream: boolean; ageRestricted: boolean; + platform: Platform; + webpageUrl: string; cues?: Cue[]; }; @@ -48,6 +58,8 @@ type Detail = { description?: string; isLivestream: boolean; ageRestricted: boolean; + platform: Platform; + webpageUrl: string; cues?: Cue[]; }; @@ -74,6 +86,8 @@ type PlayerState = { markClipEnd: () => void; clearClip: () => void; copyDownloadCommand: () => Promise<boolean>; + copyShareUrl: () => Promise<boolean>; + openInPreservetube: () => void; seekTo: (seconds: number) => void; }; @@ -111,7 +125,7 @@ export function PlayerProvider({ }); const [playing, setPlaying] = useState(false); const [currentTime, setCurrentTime] = useState(0); - const playerRef = useRef<ReactPlayerType | null>(null); + const playerRef = useRef<PlayerHandle | null>(null); const pendingSeekRef = useRef<number | null>(null); const readyForSlugRef = useRef<string | null>(null); @@ -133,6 +147,8 @@ export function PlayerProvider({ description: detail.description, isLivestream: detail.isLivestream, ageRestricted: detail.ageRestricted, + platform: detail.platform, + webpageUrl: detail.webpageUrl, cues: detail.cues, }; }, [detail, detailMatches]); @@ -151,6 +167,7 @@ export function PlayerProvider({ const t = typeof start === "number" ? Math.round(start) : null; if (slug === activeSlug && t !== null && playerRef.current) { playerRef.current.seekTo(t, "seconds"); + setCurrentTime(t); setPlaying(true); } writeUrlParams({ v: slug, t }); @@ -195,7 +212,7 @@ export function PlayerProvider({ const copyDownloadCommand = useCallback(async (): Promise<boolean> => { if (!data) return false; - const url = `https://www.youtube.com/watch?v=${data.id}`; + const url = data.webpageUrl; const sections = clipStart !== null && clipEnd !== null ? `--download-sections "*${toHMS(clipStart)}-${toHMS(clipEnd)}" ` @@ -209,6 +226,31 @@ export function PlayerProvider({ } }, [data, clipStart, clipEnd]); + const copyShareUrl = useCallback(async (): Promise<boolean> => { + if (!data) return false; + const t = data.platform === "rumble" ? (urlTime ?? 0) : currentTime; + const secs = Math.max(0, Math.floor(t)); + const params = new URLSearchParams(); + params.set("v", data.slug); + if (secs > 0) params.set("t", String(secs)); + const url = `${window.location.origin}${window.location.pathname}?${params.toString()}`; + try { + await navigator.clipboard.writeText(url); + return true; + } catch { + return false; + } + }, [data, currentTime, urlTime]); + + const openInPreservetube = useCallback(() => { + if (!data || data.platform !== "youtube") return; + window.open( + `https://preservetube.com/watch?v=${encodeURIComponent(data.id)}`, + "_blank", + "noopener,noreferrer", + ); + }, [data]); + const seekTo = useCallback((seconds: number) => { if (playerRef.current) { playerRef.current.seekTo(seconds, "seconds"); @@ -216,6 +258,7 @@ export function PlayerProvider({ } else { pendingSeekRef.current = seconds; } + setCurrentTime(seconds); }, []); useEffect(() => { @@ -236,6 +279,8 @@ export function PlayerProvider({ description: full.description, isLivestream: full.isLivestream, ageRestricted: full.ageRestricted, + platform: full.platform, + webpageUrl: full.webpageUrl, cues: full.cues, }); }) @@ -254,8 +299,22 @@ export function PlayerProvider({ if (urlTime === null) return; if (readyForSlugRef.current !== urlSlug) return; playerRef.current?.seekTo(urlTime, "seconds"); + setCurrentTime(urlTime); }, [urlSlug, urlTime]); + // Rumble has no onReady or progress events; seed currentTime from the URL so + // clip-mark buttons and active-cue highlighting have a baseline, and mark + // the slug ready so later ?t= changes can seek via the imperative handle. + useEffect(() => { + if (detail?.platform === "rumble") { + readyForSlugRef.current = detail.slug; + pendingSeekRef.current = null; + setCurrentTime(urlTime ?? 0); + } + // urlTime intentionally omitted — the urlTime effect above handles later changes. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [detail]); + useEffect(() => { if (!modalOpen) return; const onKey = (e: KeyboardEvent) => { @@ -295,6 +354,8 @@ export function PlayerProvider({ markClipEnd, clearClip, copyDownloadCommand, + copyShareUrl, + openInPreservetube, seekTo, }; @@ -320,22 +381,33 @@ export function PlayerProvider({ {showPlayer && ( <div className={containerClass} aria-hidden={displayMode === "hidden"}> - <ReactPlayer - ref={(p: ReactPlayerType | null) => { - playerRef.current = p; - }} - url={`https://www.youtube.com/watch?v=${data.id}`} - width="100%" - height="100%" - playing={playing} - controls - onReady={handleReady} - onPlay={() => setPlaying(true)} - onPause={() => setPlaying(false)} - onProgress={(s: { playedSeconds: number }) => - setCurrentTime(s.playedSeconds) - } - /> + {data.platform === "rumble" ? ( + <RumblePlayer + key={data.id} + ref={(p: RumblePlayerHandle | null) => { + playerRef.current = p; + }} + videoId={data.id} + startSeconds={urlTime ?? 0} + /> + ) : ( + <ReactPlayer + ref={(p: ReactPlayerType | null) => { + playerRef.current = p; + }} + url={`https://www.youtube.com/watch?v=${data.id}`} + width="100%" + height="100%" + playing={playing} + controls + onReady={handleReady} + onPlay={() => setPlaying(true)} + onPause={() => setPlaying(false)} + onProgress={(s: { playedSeconds: number }) => + setCurrentTime(s.playedSeconds) + } + /> + )} {displayMode === "mini" && ( <button type="button" diff --git a/app/RumblePlayer.tsx b/app/RumblePlayer.tsx @@ -0,0 +1,47 @@ +"use client"; + +import { forwardRef, useImperativeHandle, useState } from "react"; + +export type RumblePlayerHandle = { + seekTo: (seconds: number, unit?: "seconds") => void; +}; + +type Props = { + videoId: string; + startSeconds?: number; +}; + +// Rumble's iframe embed — reloading with a new `start` query param is our only +// way to seek from the outside (no JS API, no currentTime tracking). +const RumblePlayer = forwardRef<RumblePlayerHandle, Props>(function RumblePlayer( + { videoId, startSeconds = 0 }, + ref, +) { + const [start, setStart] = useState(() => + Math.max(0, Math.floor(startSeconds)), + ); + + useImperativeHandle( + ref, + () => ({ + seekTo(seconds: number) { + setStart(Math.max(0, Math.floor(seconds))); + }, + }), + [], + ); + + const src = `https://rumble.com/embed/${encodeURIComponent(videoId)}/?pub=4&start=${start}`; + + return ( + <iframe + key={start} + src={src} + allow="autoplay; fullscreen; encrypted-media; picture-in-picture" + allowFullScreen + style={{ border: 0, width: "100%", height: "100%" }} + /> + ); +}); + +export default RumblePlayer; diff --git a/app/TranscriptModal.tsx b/app/TranscriptModal.tsx @@ -21,12 +21,16 @@ export default function TranscriptModal() { markClipEnd, clearClip, copyDownloadCommand, + copyShareUrl, + openInPreservetube, seekTo, } = usePlayer(); const activeRef = useRef<HTMLLIElement | null>(null); const scrollRef = useRef<HTMLDivElement | null>(null); const [copied, setCopied] = useState(false); const copyResetRef = useRef<number | null>(null); + const [shareCopied, setShareCopied] = useState(false); + const shareResetRef = useRef<number | null>(null); const cues = data?.cues ?? []; const activeIndex = findActiveIndex(cues, currentTime); @@ -49,6 +53,7 @@ export default function TranscriptModal() { useEffect(() => { return () => { if (copyResetRef.current !== null) window.clearTimeout(copyResetRef.current); + if (shareResetRef.current !== null) window.clearTimeout(shareResetRef.current); }; }, []); @@ -62,6 +67,13 @@ export default function TranscriptModal() { if (copyResetRef.current !== null) window.clearTimeout(copyResetRef.current); copyResetRef.current = window.setTimeout(() => setCopied(false), 1800); }; + const onShare = async () => { + const ok = await copyShareUrl(); + if (!ok) return; + setShareCopied(true); + if (shareResetRef.current !== null) window.clearTimeout(shareResetRef.current); + shareResetRef.current = window.setTimeout(() => setShareCopied(false), 1800); + }; return ( <div className="fixed inset-0 z-50 flex flex-col"> @@ -112,6 +124,25 @@ export default function TranscriptModal() { /> <div className="flex-1" /> <ControlButton + title={ + !data + ? "Loading…" + : shareCopied + ? "Copied!" + : "Copy share link at current time" + } + onClick={onShare} + char={shareCopied ? "✓" : "⤴"} + disabled={!data} + /> + {data?.platform === "youtube" && ( + <ControlButton + title="Open in Preservetube" + onClick={openInPreservetube} + char="⧉" + /> + )} + <ControlButton title="Collapse to mini-player" onClick={() => setDisplayMode("mini")} char="↘" diff --git a/app/summariesCache.ts b/app/summariesCache.ts @@ -46,13 +46,21 @@ export function useSummaries(): SummariesState { const summariesReady = pageCount > 0 && loadedPages === pageCount; // Concatenate available pages; sort once all pages are in. Pages may arrive - // out of order, so a late sort keeps the slug-descending (newest-first) - // ordering stable regardless of arrival order. + // out of order, so a late sort keeps the newest-first ordering stable + // regardless of arrival order. Matches the LMDB composite-key order + // [uploadDate, channelSlug, id] reversed. const summaries = useMemo<DisplaySummary[]>(() => { if (loadedPages === 0) return []; const out: DisplaySummary[] = []; for (const q of pageQueries) if (q.data) out.push(...q.data); - if (summariesReady) out.sort((a, b) => b.slug.localeCompare(a.slug)); + if (summariesReady) { + out.sort( + (a, b) => + b.uploadDate.localeCompare(a.uploadDate) || + a.channelSlug.localeCompare(b.channelSlug) || + a.id.localeCompare(b.id), + ); + } return out; // pageQueries identity changes every render; key off loadedPages + ready. // eslint-disable-next-line react-hooks/exhaustive-deps diff --git a/lib/transcripts-server.ts b/lib/transcripts-server.ts @@ -1,5 +1,5 @@ import { formatDate, formatDuration } from "./format"; -import type { DisplaySummary, TranscriptSummary } from "./transcripts"; +import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; export type RawMetadata = { id?: string; @@ -13,10 +13,28 @@ export type RawMetadata = { was_live?: boolean; live_status?: string; age_limit?: number; + extractor?: string; + extractor_key?: string; + webpage_url?: string; }; -export function summarize(slug: string, meta: RawMetadata): TranscriptSummary { - const dateFromSlug = slug.match(/^(\d{8})_/)?.[1]; +function detectPlatform(meta: RawMetadata): Platform { + const key = meta.extractor_key ?? meta.extractor ?? ""; + if (/^rumble/i.test(key)) return "rumble"; + return "youtube"; +} + +function defaultWebpageUrl(platform: Platform, id: string): string { + if (platform === "rumble") return `https://rumble.com/${id}`; + return `https://www.youtube.com/watch?v=${id}`; +} + +export function summarize( + channelSlug: string, + videoDir: string, + meta: RawMetadata, + configName?: string, +): TranscriptSummary { const liveStatus = meta.live_status ?? ""; const isLivestream = meta.was_live === true || @@ -24,16 +42,22 @@ export function summarize(slug: string, meta: RawMetadata): TranscriptSummary { liveStatus === "was_live" || liveStatus === "is_live" || liveStatus === "is_upcoming"; + const id = meta.id ?? videoDir; + const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1]; + const platform = detectPlatform(meta); return { - slug, - id: meta.id ?? slug.split("_")[1]?.split("-")[0] ?? slug, - title: meta.title ?? slug, - uploadDate: meta.upload_date ?? dateFromSlug ?? "", + slug: `${channelSlug}/${id}`, + id, + channelSlug, + title: meta.title ?? id, + uploadDate: meta.upload_date ?? dateFromDir ?? "", duration: meta.duration ?? 0, - channel: meta.channel ?? meta.uploader ?? "", + channel: configName ?? meta.channel ?? meta.uploader ?? "", description: meta.description ?? "", isLivestream, ageRestricted: (meta.age_limit ?? 0) > 0, + platform, + webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id), }; } @@ -41,6 +65,7 @@ export function toDisplaySummary(t: TranscriptSummary): DisplaySummary { return { slug: t.slug, id: t.id, + channelSlug: t.channelSlug, title: t.title, uploadDate: t.uploadDate, date: formatDate(t.uploadDate), @@ -48,5 +73,7 @@ export function toDisplaySummary(t: TranscriptSummary): DisplaySummary { channel: t.channel, isLivestream: t.isLivestream, ageRestricted: t.ageRestricted, + platform: t.platform, + webpageUrl: t.webpageUrl, }; } diff --git a/lib/transcripts.ts b/lib/transcripts.ts @@ -9,9 +9,12 @@ const MANIFEST_PATH = path.join( "manifest.json", ); +export type Platform = "youtube" | "rumble"; + export type TranscriptSummary = { slug: string; id: string; + channelSlug: string; title: string; uploadDate: string; duration: number; @@ -19,11 +22,14 @@ export type TranscriptSummary = { description: string; isLivestream: boolean; ageRestricted: boolean; + platform: Platform; + webpageUrl: string; }; export type DisplaySummary = { slug: string; id: string; + channelSlug: string; title: string; uploadDate: string; date: string; @@ -31,6 +37,8 @@ export type DisplaySummary = { channel: string; isLivestream: boolean; ageRestricted: boolean; + platform: Platform; + webpageUrl: string; }; export type TranscriptDetail = TranscriptSummary & { diff --git a/lib/whisper.ts b/lib/whisper.ts @@ -0,0 +1,32 @@ +import type { Cue } from "./vtt"; + +type WhisperSegment = { + offsets?: { from?: number; to?: number }; + text?: string; +}; + +type WhisperDoc = { + transcription?: WhisperSegment[]; +}; + +export function parseWhisper(src: string): Cue[] { + const doc = JSON.parse(src) as WhisperDoc; + const segments = doc.transcription ?? []; + const cues: Cue[] = []; + for (const seg of segments) { + const fromMs = seg.offsets?.from; + const toMs = seg.offsets?.to; + if (typeof fromMs !== "number" || typeof toMs !== "number") continue; + const text = (seg.text ?? "").trim(); + if (!text) continue; + const start = fromMs / 1000; + const end = toMs / 1000; + const prev = cues[cues.length - 1]; + if (prev && prev.text === text) { + prev.end = end; + continue; + } + cues.push({ start, end, text }); + } + return cues; +} diff --git a/package.json b/package.json @@ -12,6 +12,7 @@ "lint": "eslint" }, "dependencies": { + "@sindresorhus/slugify": "^3.0.0", "@tanstack/react-query": "^5.99.1", "execa": "^9.6.1", "lmdb": "^3.5.4", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -8,6 +8,9 @@ importers: .: dependencies: + '@sindresorhus/slugify': + specifier: ^3.0.0 + version: 3.0.0 '@tanstack/react-query': specifier: ^5.99.1 version: 5.99.1(react@19.2.4) @@ -687,6 +690,14 @@ packages: resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==} engines: {node: '>=18'} + '@sindresorhus/slugify@3.0.0': + resolution: {integrity: sha512-SCrKh1zS96q+CuH5GumHcyQEVPsM4Ve8oE0E6tw7AAhGq50K8ojbTUOQnX/j9Mhcv/AXiIsbCfquovyGOo5fGw==} + engines: {node: '>=20'} + + '@sindresorhus/transliterate@2.3.1': + resolution: {integrity: sha512-gVaaGtKYMYAMmI8buULVH3A2TXVJ98QiwGwI7ddrWGuGidGC2uRt4FHs22+8iROJ0QTzju9CuMjlVsrvpqsdhA==} + engines: {node: '>=20'} + '@swc/helpers@0.5.15': resolution: {integrity: sha512-JQ5TuMi45Owi4/BIMAJBoSQoOJu12oOk/gADqlcUL9JEdHB8vyjUSsxqeNXnmXHjYKMi2WcYtezGEEhqUI/E2g==} @@ -1333,6 +1344,10 @@ packages: resolution: {integrity: sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA==} engines: {node: '>=10'} + escape-string-regexp@5.0.0: + resolution: {integrity: sha512-/veY75JbMK4j1yjvuUxuVsiS/hr/4iHs9FTT6cgTexxdE0Ly/glccBAkloH/DofkjRbZU3bnoj38mOmhkZ0lHw==} + engines: {node: '>=12'} + eslint-config-next@16.2.3: resolution: {integrity: sha512-Dnkrylzjof/Az7iNoIQJqD18zTxQZcngir19KJaiRsMnnjpQSVoa6aEg/1Q4hQC+cW90uTlgQYadwL1CYNwFWA==} peerDependencies: @@ -3076,6 +3091,13 @@ snapshots: '@sindresorhus/merge-streams@4.0.0': {} + '@sindresorhus/slugify@3.0.0': + dependencies: + '@sindresorhus/transliterate': 2.3.1 + escape-string-regexp: 5.0.0 + + '@sindresorhus/transliterate@2.3.1': {} + '@swc/helpers@0.5.15': dependencies: tslib: 2.8.1 @@ -3792,6 +3814,8 @@ snapshots: escape-string-regexp@4.0.0: {} + escape-string-regexp@5.0.0: {} + eslint-config-next@16.2.3(@typescript-eslint/parser@8.58.2(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3))(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3): dependencies: '@next/eslint-plugin-next': 16.2.3 diff --git a/scripts/build-index.ts b/scripts/build-index.ts @@ -1,38 +1,74 @@ #!/usr/bin/env tsx -// Preprocess transcripts/data/* into: +// Preprocess transcripts/channels/<channelSlug>/data/<videoDir>/ into: // - LMDB cache at transcripts/index.mdb (for incremental rebuilds) // - public/summaries/{manifest,page-NNNN}.json (paginated summaries index) -// - public/transcripts/<slug>.json (per-transcript cues + metadata) +// - public/transcripts/<channelSlug>/<id>.json (per-transcript cues + metadata) // -// Short-circuits when mtimes already match and the public outputs are intact. +// Per-channel config.json selects the transcript parser ("youtube" → VTT, +// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match. import path from "node:path"; -import { mkdir, readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises"; +import { + mkdir, + readdir, + readFile, + rename, + rm, + stat, + writeFile, +} from "node:fs/promises"; +import type { Dirent } from "node:fs"; import { open } from "lmdb"; import { parseVtt, type Cue } from "../lib/vtt"; -import { summarize, toDisplaySummary, type RawMetadata } from "../lib/transcripts-server"; +import { parseWhisper } from "../lib/whisper"; +import { + summarize, + toDisplaySummary, + type RawMetadata, +} from "../lib/transcripts-server"; import type { TranscriptSummary, DisplaySummary } from "../lib/transcripts"; import type { Manifest, ChannelEntry } from "../lib/manifest"; -import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE, pageFileName } from "../lib/manifest"; +import { + MANIFEST_VERSION, + SUMMARIES_PAGE_SIZE, + pageFileName, +} from "../lib/manifest"; const ROOT = process.cwd(); -const DATA_DIR = path.join(ROOT, "transcripts", "data"); +const CHANNELS_DIR = path.join(ROOT, "transcripts", "channels"); const DB_PATH = path.join(ROOT, "transcripts", "index.mdb"); const PUBLIC_DIR = path.join(ROOT, "public"); const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries"); const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts"); const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json"); -const SCHEMA_VERSION = 1; +const SCHEMA_VERSION = 3; -type MtimeRecord = { metaMs: number; vttMs: number | null }; +type Handling = "youtube" | "transcribe"; +type ChannelConfig = { handling: Handling; name?: string }; -type LiveSlug = { - slug: string; +// Primary sort/lookup key: [uploadDate, channelSlug, id]. Reverse iteration +// yields newest-first directly (uploadDate is YYYYMMDD). +type IndexKey = [string, string, string]; +// Path key: [channelSlug, videoDir]. Stable regardless of metadata contents, +// so diffing by mtime doesn't require parsing metadata.info.json. +type PathKey = [string, string]; + +type MtimeRecord = { + metaMs: number; + transcriptMs: number | null; + indexKey: IndexKey; +}; + +type LiveEntry = { + channelSlug: string; + handling: Handling; + configName: string | undefined; + videoDir: string; metaPath: string; metaMs: number; - vttPath: string; - vttMs: number | null; + transcriptPath: string; + transcriptMs: number | null; }; async function exists(p: string): Promise<boolean> { @@ -44,33 +80,90 @@ async function exists(p: string): Promise<boolean> { } } -async function scanSource(): Promise<LiveSlug[]> { - const entries = await readdir(DATA_DIR, { withFileTypes: true }); - const out: LiveSlug[] = []; - for (const e of entries) { - if (!e.isDirectory()) continue; - const slug = e.name; - const metaPath = path.join(DATA_DIR, slug, "metadata.info.json"); - const vttPath = path.join(DATA_DIR, slug, "transcript.en.vtt"); - let metaMs: number; - try { - metaMs = (await stat(metaPath)).mtimeMs; - } catch { - // Skip slugs without metadata — they're not usable. +async function readChannelConfig(dir: string): Promise<ChannelConfig | null> { + try { + const raw = await readFile(path.join(dir, "config.json"), "utf8"); + const parsed = JSON.parse(raw) as ChannelConfig; + if (parsed.handling !== "youtube" && parsed.handling !== "transcribe") { + return null; + } + return parsed; + } catch { + return null; + } +} + +async function scanSource(): Promise<{ + live: LiveEntry[]; + channels: Map<string, ChannelConfig>; +}> { + const channels = new Map<string, ChannelConfig>(); + const live: LiveEntry[] = []; + const channelEntries = await readdir(CHANNELS_DIR, { withFileTypes: true }); + for (const ch of channelEntries) { + if (!ch.isDirectory()) continue; + const channelDir = path.join(CHANNELS_DIR, ch.name); + const cfg = await readChannelConfig(channelDir); + if (!cfg) { + console.warn( + `Skipping channel ${ch.name}: missing or invalid config.json`, + ); continue; } - let vttMs: number | null = null; + channels.set(ch.name, cfg); + const dataDir = path.join(channelDir, "data"); + let videoEntries: Dirent[]; try { - vttMs = (await stat(vttPath)).mtimeMs; + videoEntries = await readdir(dataDir, { withFileTypes: true }); } catch { - vttMs = null; + continue; + } + const transcriptName = + cfg.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; + for (const v of videoEntries) { + if (!v.isDirectory()) continue; + const videoDir = v.name; + const metaPath = path.join(dataDir, videoDir, "metadata.info.json"); + const transcriptPath = path.join(dataDir, videoDir, transcriptName); + let metaMs: number; + try { + metaMs = (await stat(metaPath)).mtimeMs; + } catch { + continue; + } + let transcriptMs: number | null = null; + try { + transcriptMs = (await stat(transcriptPath)).mtimeMs; + } catch { + transcriptMs = null; + } + live.push({ + channelSlug: ch.name, + handling: cfg.handling, + configName: cfg.name, + videoDir, + metaPath, + metaMs, + transcriptPath, + transcriptMs, + }); } - out.push({ slug, metaPath, metaMs, vttPath, vttMs }); } - return out; + return { live, channels }; } -async function writeJsonAtomic(filePath: string, value: unknown): Promise<void> { +function pathKeyId(k: PathKey): string { + return `${k[0]}\x00${k[1]}`; +} + +function indexKeysEqual(a: IndexKey, b: IndexKey): boolean { + return a[0] === b[0] && a[1] === b[1] && a[2] === b[2]; +} + +async function writeJsonAtomic( + filePath: string, + value: unknown, +): Promise<void> { const tmp = `${filePath}.tmp-${process.pid}`; await writeFile(tmp, JSON.stringify(value)); await rename(tmp, filePath); @@ -87,15 +180,15 @@ async function main(): Promise<void> { maxDbs: 8, compression: true, }); - const sums = root.openDB<TranscriptSummary, string>({ + const sums = root.openDB<TranscriptSummary, IndexKey>({ name: "sums", encoding: "msgpack", }); - const cues = root.openDB<Cue[], string>({ + const cues = root.openDB<Cue[], IndexKey>({ name: "cues", encoding: "msgpack", }); - const mtimes = root.openDB<MtimeRecord, string>({ + const mtimes = root.openDB<MtimeRecord, PathKey>({ name: "mtimes", encoding: "msgpack", }); @@ -104,7 +197,6 @@ async function main(): Promise<void> { encoding: "msgpack", }); - // Force full rebuild if the schema changed. const storedSchema = meta.get("schema") as number | undefined; const schemaBumped = storedSchema !== SCHEMA_VERSION; if (schemaBumped) { @@ -117,32 +209,43 @@ async function main(): Promise<void> { await meta.put("schema", SCHEMA_VERSION); } - const live = await scanSource(); - const liveBySlug = new Map(live.map((s) => [s.slug, s])); - - const known = new Set<string>(); - for (const { key } of mtimes.getRange()) known.add(key); + const { live, channels: channelConfigs } = await scanSource(); + const livePathIds = new Set<string>(); + const liveByPathId = new Map<string, LiveEntry>(); + for (const s of live) { + const id = pathKeyId([s.channelSlug, s.videoDir]); + livePathIds.add(id); + liveByPathId.set(id, s); + } - const added: LiveSlug[] = []; - const changed: LiveSlug[] = []; - const removed: string[] = []; + // Diff against prior mtimes (path-keyed — no metadata reads required). + const added: LiveEntry[] = []; + const changed: LiveEntry[] = []; + const removed: { pathKey: PathKey; indexKey: IndexKey }[] = []; for (const s of live) { - const prev = mtimes.get(s.slug); + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); if (!prev) { added.push(s); - } else if (prev.metaMs !== s.metaMs || prev.vttMs !== s.vttMs) { + } else if ( + prev.metaMs !== s.metaMs || + prev.transcriptMs !== s.transcriptMs + ) { changed.push(s); } } - for (const slug of known) { - if (!liveBySlug.has(slug)) removed.push(slug); + for (const { key, value } of mtimes.getRange()) { + const k = key as PathKey; + if (!livePathIds.has(pathKeyId(k))) { + removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey }); + } } const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; - // Short-circuit: if nothing changed AND public/ is intact, we're done. + // Short-circuit: nothing changed AND public/ is intact. if (!anyMutations && !schemaBumped) { const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null); if (manifestRaw) { @@ -153,7 +256,6 @@ async function main(): Promise<void> { parsed.totalCount === live.length && parsed.pageSize === SUMMARIES_PAGE_SIZE ) { - // Spot-check first and last page files exist. const firstPage = path.join(SUMMARIES_DIR, pageFileName(0)); const lastPage = path.join( SUMMARIES_DIR, @@ -168,7 +270,7 @@ async function main(): Promise<void> { } } } catch { - // fall through to full emission + // fall through } } } @@ -177,7 +279,12 @@ async function main(): Promise<void> { `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, ); - // Process mutations: parse files and write to LMDB + per-slug public JSON. + // Ensure per-channel public/transcripts/<channelSlug>/ subdirs exist. + for (const channelSlug of channelConfigs.keys()) { + await mkdir(path.join(TRANSCRIPTS_DIR, channelSlug), { recursive: true }); + } + + // Process mutations. const toProcess = [...added, ...changed]; let processed = 0; const BATCH = 200; @@ -188,28 +295,62 @@ async function main(): Promise<void> { try { const metaRaw = await readFile(s.metaPath, "utf8"); const parsedMeta = JSON.parse(metaRaw) as RawMetadata; - const summary = summarize(s.slug, parsedMeta); + const summary = summarize( + s.channelSlug, + s.videoDir, + parsedMeta, + s.configName, + ); + if (!summary.uploadDate) { + console.warn( + `Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`, + ); + return; + } + const indexKey: IndexKey = [ + summary.uploadDate, + s.channelSlug, + summary.id, + ]; let cueList: Cue[] | undefined; - if (s.vttMs !== null) { + if (s.transcriptMs !== null) { try { - const vtt = await readFile(s.vttPath, "utf8"); - cueList = parseVtt(vtt); + const raw = await readFile(s.transcriptPath, "utf8"); + cueList = + s.handling === "youtube" ? parseVtt(raw) : parseWhisper(raw); } catch { cueList = undefined; } } - sums.put(s.slug, summary); - if (cueList) cues.put(s.slug, cueList); - else cues.remove(s.slug); - mtimes.put(s.slug, { metaMs: s.metaMs, vttMs: s.vttMs }); + + // If upload_date shifted (metadata edit), the old composite key is + // stale — drop it before writing the new one. + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); + if (prev && !indexKeysEqual(prev.indexKey, indexKey)) { + sums.remove(prev.indexKey); + cues.remove(prev.indexKey); + } + + sums.put(indexKey, summary); + if (cueList) cues.put(indexKey, cueList); + else cues.remove(indexKey); + mtimes.put(pk, { + metaMs: s.metaMs, + transcriptMs: s.transcriptMs, + indexKey, + }); const detail = { ...summary, cues: cueList }; await writeJsonAtomic( - path.join(TRANSCRIPTS_DIR, `${s.slug}.json`), + path.join(TRANSCRIPTS_DIR, `${summary.slug}.json`), detail, ); } catch (err) { - console.warn(`Failed to process ${s.slug}:`, err); + console.warn( + `Failed to process ${s.channelSlug}/${s.videoDir}:`, + err, + ); } }), ); @@ -221,22 +362,26 @@ async function main(): Promise<void> { if (toProcess.length > BATCH) process.stdout.write("\n"); // Process removals. - for (const slug of removed) { - sums.remove(slug); - cues.remove(slug); - mtimes.remove(slug); + for (const { pathKey, indexKey } of removed) { + sums.remove(indexKey); + cues.remove(indexKey); + mtimes.remove(pathKey); + const slug = `${indexKey[1]}/${indexKey[2]}`; await rm(path.join(TRANSCRIPTS_DIR, `${slug}.json`), { force: true }); } - // Flush writes before we iterate for emission. await sums.flushed; await cues.flushed; await mtimes.flushed; - // Stream LMDB (reverse slug order = newest first) to emit paginated summaries. - // Hold at most one page worth of summaries in memory at once. + // Stream LMDB (reverse composite-key order = newest-date first) to emit + // paginated summaries. const pageSize = SUMMARIES_PAGE_SIZE; - const channels = new Map<string, number>(); + const channelCounts = new Map<string, number>(); + // Seed channel list from config so empty channels still appear in the UI. + for (const cfg of channelConfigs.values()) { + if (cfg.name) channelCounts.set(cfg.name, 0); + } let pageIndex = 0; let buffer: DisplaySummary[] = []; let total = 0; @@ -249,18 +394,18 @@ async function main(): Promise<void> { pageIndex++; }; - // LMDB returns keys in ascending order; we want newest-first (descending slug). - // Slugs are `YYYYMMDD_...` so lexicographic reverse == chronological reverse. for (const { value } of sums.getRange({ reverse: true })) { const s = value as TranscriptSummary; - if (s.channel) channels.set(s.channel, (channels.get(s.channel) ?? 0) + 1); + if (s.channel) { + channelCounts.set(s.channel, (channelCounts.get(s.channel) ?? 0) + 1); + } buffer.push(toDisplaySummary(s)); total++; if (buffer.length >= pageSize) await flushPage(); } await flushPage(); - // Prune stale page files (if page count shrank). + // Prune stale page files. const expectedPages = new Set<string>(); for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i)); const existing = await readdir(SUMMARIES_DIR).catch(() => [] as string[]); @@ -271,7 +416,7 @@ async function main(): Promise<void> { } } - const channelList: ChannelEntry[] = Array.from(channels.entries()) + const channelList: ChannelEntry[] = Array.from(channelCounts.entries()) .map(([name, count]) => ({ name, count })) .sort((a, b) => a.name.localeCompare(b.name));