commit 552a70c17bb05455caa3e21be7dd7e89eff79343
parent 19defd08da573ceab111e94fa67f4a32e029cb05
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 21 Apr 2026 07:38:40 -0400
Add Rumble support
Diffstat:
10 files changed, 499 insertions(+), 104 deletions(-)
diff --git a/app/PlayerProvider.tsx b/app/PlayerProvider.tsx
@@ -14,11 +14,19 @@ import type ReactPlayerType from "react-player";
import { fetchTranscript } from "./transcriptCache";
import { useUrlParams, writeUrlParams } from "./urlState";
import { formatDate, formatDuration } from "@/lib/format";
+import type { Platform } from "@/lib/transcripts";
+import type { RumblePlayerHandle } from "./RumblePlayer";
const ReactPlayer = dynamic(() => import("react-player/youtube"), {
ssr: false,
});
+const RumblePlayer = dynamic(() => import("./RumblePlayer"), {
+ ssr: false,
+});
+
+type PlayerHandle = Pick<ReactPlayerType, "seekTo"> | RumblePlayerHandle;
+
export type Cue = { start: number; end: number; text: string };
export type TranscriptData = {
@@ -32,6 +40,8 @@ export type TranscriptData = {
description?: string;
isLivestream: boolean;
ageRestricted: boolean;
+ platform: Platform;
+ webpageUrl: string;
cues?: Cue[];
};
@@ -48,6 +58,8 @@ type Detail = {
description?: string;
isLivestream: boolean;
ageRestricted: boolean;
+ platform: Platform;
+ webpageUrl: string;
cues?: Cue[];
};
@@ -74,6 +86,8 @@ type PlayerState = {
markClipEnd: () => void;
clearClip: () => void;
copyDownloadCommand: () => Promise<boolean>;
+ copyShareUrl: () => Promise<boolean>;
+ openInPreservetube: () => void;
seekTo: (seconds: number) => void;
};
@@ -111,7 +125,7 @@ export function PlayerProvider({
});
const [playing, setPlaying] = useState(false);
const [currentTime, setCurrentTime] = useState(0);
- const playerRef = useRef<ReactPlayerType | null>(null);
+ const playerRef = useRef<PlayerHandle | null>(null);
const pendingSeekRef = useRef<number | null>(null);
const readyForSlugRef = useRef<string | null>(null);
@@ -133,6 +147,8 @@ export function PlayerProvider({
description: detail.description,
isLivestream: detail.isLivestream,
ageRestricted: detail.ageRestricted,
+ platform: detail.platform,
+ webpageUrl: detail.webpageUrl,
cues: detail.cues,
};
}, [detail, detailMatches]);
@@ -151,6 +167,7 @@ export function PlayerProvider({
const t = typeof start === "number" ? Math.round(start) : null;
if (slug === activeSlug && t !== null && playerRef.current) {
playerRef.current.seekTo(t, "seconds");
+ setCurrentTime(t);
setPlaying(true);
}
writeUrlParams({ v: slug, t });
@@ -195,7 +212,7 @@ export function PlayerProvider({
const copyDownloadCommand = useCallback(async (): Promise<boolean> => {
if (!data) return false;
- const url = `https://www.youtube.com/watch?v=${data.id}`;
+ const url = data.webpageUrl;
const sections =
clipStart !== null && clipEnd !== null
? `--download-sections "*${toHMS(clipStart)}-${toHMS(clipEnd)}" `
@@ -209,6 +226,31 @@ export function PlayerProvider({
}
}, [data, clipStart, clipEnd]);
+ const copyShareUrl = useCallback(async (): Promise<boolean> => {
+ if (!data) return false;
+ const t = data.platform === "rumble" ? (urlTime ?? 0) : currentTime;
+ const secs = Math.max(0, Math.floor(t));
+ const params = new URLSearchParams();
+ params.set("v", data.slug);
+ if (secs > 0) params.set("t", String(secs));
+ const url = `${window.location.origin}${window.location.pathname}?${params.toString()}`;
+ try {
+ await navigator.clipboard.writeText(url);
+ return true;
+ } catch {
+ return false;
+ }
+ }, [data, currentTime, urlTime]);
+
+ const openInPreservetube = useCallback(() => {
+ if (!data || data.platform !== "youtube") return;
+ window.open(
+ `https://preservetube.com/watch?v=${encodeURIComponent(data.id)}`,
+ "_blank",
+ "noopener,noreferrer",
+ );
+ }, [data]);
+
const seekTo = useCallback((seconds: number) => {
if (playerRef.current) {
playerRef.current.seekTo(seconds, "seconds");
@@ -216,6 +258,7 @@ export function PlayerProvider({
} else {
pendingSeekRef.current = seconds;
}
+ setCurrentTime(seconds);
}, []);
useEffect(() => {
@@ -236,6 +279,8 @@ export function PlayerProvider({
description: full.description,
isLivestream: full.isLivestream,
ageRestricted: full.ageRestricted,
+ platform: full.platform,
+ webpageUrl: full.webpageUrl,
cues: full.cues,
});
})
@@ -254,8 +299,22 @@ export function PlayerProvider({
if (urlTime === null) return;
if (readyForSlugRef.current !== urlSlug) return;
playerRef.current?.seekTo(urlTime, "seconds");
+ setCurrentTime(urlTime);
}, [urlSlug, urlTime]);
+ // Rumble has no onReady or progress events; seed currentTime from the URL so
+ // clip-mark buttons and active-cue highlighting have a baseline, and mark
+ // the slug ready so later ?t= changes can seek via the imperative handle.
+ useEffect(() => {
+ if (detail?.platform === "rumble") {
+ readyForSlugRef.current = detail.slug;
+ pendingSeekRef.current = null;
+ setCurrentTime(urlTime ?? 0);
+ }
+ // urlTime intentionally omitted — the urlTime effect above handles later changes.
+ // eslint-disable-next-line react-hooks/exhaustive-deps
+ }, [detail]);
+
useEffect(() => {
if (!modalOpen) return;
const onKey = (e: KeyboardEvent) => {
@@ -295,6 +354,8 @@ export function PlayerProvider({
markClipEnd,
clearClip,
copyDownloadCommand,
+ copyShareUrl,
+ openInPreservetube,
seekTo,
};
@@ -320,22 +381,33 @@ export function PlayerProvider({
{showPlayer && (
<div className={containerClass} aria-hidden={displayMode === "hidden"}>
- <ReactPlayer
- ref={(p: ReactPlayerType | null) => {
- playerRef.current = p;
- }}
- url={`https://www.youtube.com/watch?v=${data.id}`}
- width="100%"
- height="100%"
- playing={playing}
- controls
- onReady={handleReady}
- onPlay={() => setPlaying(true)}
- onPause={() => setPlaying(false)}
- onProgress={(s: { playedSeconds: number }) =>
- setCurrentTime(s.playedSeconds)
- }
- />
+ {data.platform === "rumble" ? (
+ <RumblePlayer
+ key={data.id}
+ ref={(p: RumblePlayerHandle | null) => {
+ playerRef.current = p;
+ }}
+ videoId={data.id}
+ startSeconds={urlTime ?? 0}
+ />
+ ) : (
+ <ReactPlayer
+ ref={(p: ReactPlayerType | null) => {
+ playerRef.current = p;
+ }}
+ url={`https://www.youtube.com/watch?v=${data.id}`}
+ width="100%"
+ height="100%"
+ playing={playing}
+ controls
+ onReady={handleReady}
+ onPlay={() => setPlaying(true)}
+ onPause={() => setPlaying(false)}
+ onProgress={(s: { playedSeconds: number }) =>
+ setCurrentTime(s.playedSeconds)
+ }
+ />
+ )}
{displayMode === "mini" && (
<button
type="button"
diff --git a/app/RumblePlayer.tsx b/app/RumblePlayer.tsx
@@ -0,0 +1,47 @@
+"use client";
+
+import { forwardRef, useImperativeHandle, useState } from "react";
+
+export type RumblePlayerHandle = {
+ seekTo: (seconds: number, unit?: "seconds") => void;
+};
+
+type Props = {
+ videoId: string;
+ startSeconds?: number;
+};
+
+// Rumble's iframe embed — reloading with a new `start` query param is our only
+// way to seek from the outside (no JS API, no currentTime tracking).
+const RumblePlayer = forwardRef<RumblePlayerHandle, Props>(function RumblePlayer(
+ { videoId, startSeconds = 0 },
+ ref,
+) {
+ const [start, setStart] = useState(() =>
+ Math.max(0, Math.floor(startSeconds)),
+ );
+
+ useImperativeHandle(
+ ref,
+ () => ({
+ seekTo(seconds: number) {
+ setStart(Math.max(0, Math.floor(seconds)));
+ },
+ }),
+ [],
+ );
+
+ const src = `https://rumble.com/embed/${encodeURIComponent(videoId)}/?pub=4&start=${start}`;
+
+ return (
+ <iframe
+ key={start}
+ src={src}
+ allow="autoplay; fullscreen; encrypted-media; picture-in-picture"
+ allowFullScreen
+ style={{ border: 0, width: "100%", height: "100%" }}
+ />
+ );
+});
+
+export default RumblePlayer;
diff --git a/app/TranscriptModal.tsx b/app/TranscriptModal.tsx
@@ -21,12 +21,16 @@ export default function TranscriptModal() {
markClipEnd,
clearClip,
copyDownloadCommand,
+ copyShareUrl,
+ openInPreservetube,
seekTo,
} = usePlayer();
const activeRef = useRef<HTMLLIElement | null>(null);
const scrollRef = useRef<HTMLDivElement | null>(null);
const [copied, setCopied] = useState(false);
const copyResetRef = useRef<number | null>(null);
+ const [shareCopied, setShareCopied] = useState(false);
+ const shareResetRef = useRef<number | null>(null);
const cues = data?.cues ?? [];
const activeIndex = findActiveIndex(cues, currentTime);
@@ -49,6 +53,7 @@ export default function TranscriptModal() {
useEffect(() => {
return () => {
if (copyResetRef.current !== null) window.clearTimeout(copyResetRef.current);
+ if (shareResetRef.current !== null) window.clearTimeout(shareResetRef.current);
};
}, []);
@@ -62,6 +67,13 @@ export default function TranscriptModal() {
if (copyResetRef.current !== null) window.clearTimeout(copyResetRef.current);
copyResetRef.current = window.setTimeout(() => setCopied(false), 1800);
};
+ const onShare = async () => {
+ const ok = await copyShareUrl();
+ if (!ok) return;
+ setShareCopied(true);
+ if (shareResetRef.current !== null) window.clearTimeout(shareResetRef.current);
+ shareResetRef.current = window.setTimeout(() => setShareCopied(false), 1800);
+ };
return (
<div className="fixed inset-0 z-50 flex flex-col">
@@ -112,6 +124,25 @@ export default function TranscriptModal() {
/>
<div className="flex-1" />
<ControlButton
+ title={
+ !data
+ ? "Loading…"
+ : shareCopied
+ ? "Copied!"
+ : "Copy share link at current time"
+ }
+ onClick={onShare}
+ char={shareCopied ? "✓" : "⤴"}
+ disabled={!data}
+ />
+ {data?.platform === "youtube" && (
+ <ControlButton
+ title="Open in Preservetube"
+ onClick={openInPreservetube}
+ char="⧉"
+ />
+ )}
+ <ControlButton
title="Collapse to mini-player"
onClick={() => setDisplayMode("mini")}
char="↘"
diff --git a/app/summariesCache.ts b/app/summariesCache.ts
@@ -46,13 +46,21 @@ export function useSummaries(): SummariesState {
const summariesReady = pageCount > 0 && loadedPages === pageCount;
// Concatenate available pages; sort once all pages are in. Pages may arrive
- // out of order, so a late sort keeps the slug-descending (newest-first)
- // ordering stable regardless of arrival order.
+ // out of order, so a late sort keeps the newest-first ordering stable
+ // regardless of arrival order. Matches the LMDB composite-key order
+ // [uploadDate, channelSlug, id] reversed.
const summaries = useMemo<DisplaySummary[]>(() => {
if (loadedPages === 0) return [];
const out: DisplaySummary[] = [];
for (const q of pageQueries) if (q.data) out.push(...q.data);
- if (summariesReady) out.sort((a, b) => b.slug.localeCompare(a.slug));
+ if (summariesReady) {
+ out.sort(
+ (a, b) =>
+ b.uploadDate.localeCompare(a.uploadDate) ||
+ a.channelSlug.localeCompare(b.channelSlug) ||
+ a.id.localeCompare(b.id),
+ );
+ }
return out;
// pageQueries identity changes every render; key off loadedPages + ready.
// eslint-disable-next-line react-hooks/exhaustive-deps
diff --git a/lib/transcripts-server.ts b/lib/transcripts-server.ts
@@ -1,5 +1,5 @@
import { formatDate, formatDuration } from "./format";
-import type { DisplaySummary, TranscriptSummary } from "./transcripts";
+import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
export type RawMetadata = {
id?: string;
@@ -13,10 +13,28 @@ export type RawMetadata = {
was_live?: boolean;
live_status?: string;
age_limit?: number;
+ extractor?: string;
+ extractor_key?: string;
+ webpage_url?: string;
};
-export function summarize(slug: string, meta: RawMetadata): TranscriptSummary {
- const dateFromSlug = slug.match(/^(\d{8})_/)?.[1];
+function detectPlatform(meta: RawMetadata): Platform {
+ const key = meta.extractor_key ?? meta.extractor ?? "";
+ if (/^rumble/i.test(key)) return "rumble";
+ return "youtube";
+}
+
+function defaultWebpageUrl(platform: Platform, id: string): string {
+ if (platform === "rumble") return `https://rumble.com/${id}`;
+ return `https://www.youtube.com/watch?v=${id}`;
+}
+
+export function summarize(
+ channelSlug: string,
+ videoDir: string,
+ meta: RawMetadata,
+ configName?: string,
+): TranscriptSummary {
const liveStatus = meta.live_status ?? "";
const isLivestream =
meta.was_live === true ||
@@ -24,16 +42,22 @@ export function summarize(slug: string, meta: RawMetadata): TranscriptSummary {
liveStatus === "was_live" ||
liveStatus === "is_live" ||
liveStatus === "is_upcoming";
+ const id = meta.id ?? videoDir;
+ const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
+ const platform = detectPlatform(meta);
return {
- slug,
- id: meta.id ?? slug.split("_")[1]?.split("-")[0] ?? slug,
- title: meta.title ?? slug,
- uploadDate: meta.upload_date ?? dateFromSlug ?? "",
+ slug: `${channelSlug}/${id}`,
+ id,
+ channelSlug,
+ title: meta.title ?? id,
+ uploadDate: meta.upload_date ?? dateFromDir ?? "",
duration: meta.duration ?? 0,
- channel: meta.channel ?? meta.uploader ?? "",
+ channel: configName ?? meta.channel ?? meta.uploader ?? "",
description: meta.description ?? "",
isLivestream,
ageRestricted: (meta.age_limit ?? 0) > 0,
+ platform,
+ webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id),
};
}
@@ -41,6 +65,7 @@ export function toDisplaySummary(t: TranscriptSummary): DisplaySummary {
return {
slug: t.slug,
id: t.id,
+ channelSlug: t.channelSlug,
title: t.title,
uploadDate: t.uploadDate,
date: formatDate(t.uploadDate),
@@ -48,5 +73,7 @@ export function toDisplaySummary(t: TranscriptSummary): DisplaySummary {
channel: t.channel,
isLivestream: t.isLivestream,
ageRestricted: t.ageRestricted,
+ platform: t.platform,
+ webpageUrl: t.webpageUrl,
};
}
diff --git a/lib/transcripts.ts b/lib/transcripts.ts
@@ -9,9 +9,12 @@ const MANIFEST_PATH = path.join(
"manifest.json",
);
+export type Platform = "youtube" | "rumble";
+
export type TranscriptSummary = {
slug: string;
id: string;
+ channelSlug: string;
title: string;
uploadDate: string;
duration: number;
@@ -19,11 +22,14 @@ export type TranscriptSummary = {
description: string;
isLivestream: boolean;
ageRestricted: boolean;
+ platform: Platform;
+ webpageUrl: string;
};
export type DisplaySummary = {
slug: string;
id: string;
+ channelSlug: string;
title: string;
uploadDate: string;
date: string;
@@ -31,6 +37,8 @@ export type DisplaySummary = {
channel: string;
isLivestream: boolean;
ageRestricted: boolean;
+ platform: Platform;
+ webpageUrl: string;
};
export type TranscriptDetail = TranscriptSummary & {
diff --git a/lib/whisper.ts b/lib/whisper.ts
@@ -0,0 +1,32 @@
+import type { Cue } from "./vtt";
+
+type WhisperSegment = {
+ offsets?: { from?: number; to?: number };
+ text?: string;
+};
+
+type WhisperDoc = {
+ transcription?: WhisperSegment[];
+};
+
+export function parseWhisper(src: string): Cue[] {
+ const doc = JSON.parse(src) as WhisperDoc;
+ const segments = doc.transcription ?? [];
+ const cues: Cue[] = [];
+ for (const seg of segments) {
+ const fromMs = seg.offsets?.from;
+ const toMs = seg.offsets?.to;
+ if (typeof fromMs !== "number" || typeof toMs !== "number") continue;
+ const text = (seg.text ?? "").trim();
+ if (!text) continue;
+ const start = fromMs / 1000;
+ const end = toMs / 1000;
+ const prev = cues[cues.length - 1];
+ if (prev && prev.text === text) {
+ prev.end = end;
+ continue;
+ }
+ cues.push({ start, end, text });
+ }
+ return cues;
+}
diff --git a/package.json b/package.json
@@ -12,6 +12,7 @@
"lint": "eslint"
},
"dependencies": {
+ "@sindresorhus/slugify": "^3.0.0",
"@tanstack/react-query": "^5.99.1",
"execa": "^9.6.1",
"lmdb": "^3.5.4",
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
@@ -8,6 +8,9 @@ importers:
.:
dependencies:
+ '@sindresorhus/slugify':
+ specifier: ^3.0.0
+ version: 3.0.0
'@tanstack/react-query':
specifier: ^5.99.1
version: 5.99.1(react@19.2.4)
@@ -687,6 +690,14 @@ packages:
resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==}
engines: {node: '>=18'}
+ '@sindresorhus/slugify@3.0.0':
+ resolution: {integrity: sha512-SCrKh1zS96q+CuH5GumHcyQEVPsM4Ve8oE0E6tw7AAhGq50K8ojbTUOQnX/j9Mhcv/AXiIsbCfquovyGOo5fGw==}
+ engines: {node: '>=20'}
+
+ '@sindresorhus/transliterate@2.3.1':
+ resolution: {integrity: sha512-gVaaGtKYMYAMmI8buULVH3A2TXVJ98QiwGwI7ddrWGuGidGC2uRt4FHs22+8iROJ0QTzju9CuMjlVsrvpqsdhA==}
+ engines: {node: '>=20'}
+
'@swc/helpers@0.5.15':
resolution: {integrity: sha512-JQ5TuMi45Owi4/BIMAJBoSQoOJu12oOk/gADqlcUL9JEdHB8vyjUSsxqeNXnmXHjYKMi2WcYtezGEEhqUI/E2g==}
@@ -1333,6 +1344,10 @@ packages:
resolution: {integrity: sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA==}
engines: {node: '>=10'}
+ escape-string-regexp@5.0.0:
+ resolution: {integrity: sha512-/veY75JbMK4j1yjvuUxuVsiS/hr/4iHs9FTT6cgTexxdE0Ly/glccBAkloH/DofkjRbZU3bnoj38mOmhkZ0lHw==}
+ engines: {node: '>=12'}
+
eslint-config-next@16.2.3:
resolution: {integrity: sha512-Dnkrylzjof/Az7iNoIQJqD18zTxQZcngir19KJaiRsMnnjpQSVoa6aEg/1Q4hQC+cW90uTlgQYadwL1CYNwFWA==}
peerDependencies:
@@ -3076,6 +3091,13 @@ snapshots:
'@sindresorhus/merge-streams@4.0.0': {}
+ '@sindresorhus/slugify@3.0.0':
+ dependencies:
+ '@sindresorhus/transliterate': 2.3.1
+ escape-string-regexp: 5.0.0
+
+ '@sindresorhus/transliterate@2.3.1': {}
+
'@swc/helpers@0.5.15':
dependencies:
tslib: 2.8.1
@@ -3792,6 +3814,8 @@ snapshots:
escape-string-regexp@4.0.0: {}
+ escape-string-regexp@5.0.0: {}
+
eslint-config-next@16.2.3(@typescript-eslint/parser@8.58.2(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3))(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3):
dependencies:
'@next/eslint-plugin-next': 16.2.3
diff --git a/scripts/build-index.ts b/scripts/build-index.ts
@@ -1,38 +1,74 @@
#!/usr/bin/env tsx
-// Preprocess transcripts/data/* into:
+// Preprocess transcripts/channels/<channelSlug>/data/<videoDir>/ into:
// - LMDB cache at transcripts/index.mdb (for incremental rebuilds)
// - public/summaries/{manifest,page-NNNN}.json (paginated summaries index)
-// - public/transcripts/<slug>.json (per-transcript cues + metadata)
+// - public/transcripts/<channelSlug>/<id>.json (per-transcript cues + metadata)
//
-// Short-circuits when mtimes already match and the public outputs are intact.
+// Per-channel config.json selects the transcript parser ("youtube" → VTT,
+// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match.
import path from "node:path";
-import { mkdir, readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
+import {
+ mkdir,
+ readdir,
+ readFile,
+ rename,
+ rm,
+ stat,
+ writeFile,
+} from "node:fs/promises";
+import type { Dirent } from "node:fs";
import { open } from "lmdb";
import { parseVtt, type Cue } from "../lib/vtt";
-import { summarize, toDisplaySummary, type RawMetadata } from "../lib/transcripts-server";
+import { parseWhisper } from "../lib/whisper";
+import {
+ summarize,
+ toDisplaySummary,
+ type RawMetadata,
+} from "../lib/transcripts-server";
import type { TranscriptSummary, DisplaySummary } from "../lib/transcripts";
import type { Manifest, ChannelEntry } from "../lib/manifest";
-import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE, pageFileName } from "../lib/manifest";
+import {
+ MANIFEST_VERSION,
+ SUMMARIES_PAGE_SIZE,
+ pageFileName,
+} from "../lib/manifest";
const ROOT = process.cwd();
-const DATA_DIR = path.join(ROOT, "transcripts", "data");
+const CHANNELS_DIR = path.join(ROOT, "transcripts", "channels");
const DB_PATH = path.join(ROOT, "transcripts", "index.mdb");
const PUBLIC_DIR = path.join(ROOT, "public");
const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries");
const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts");
const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json");
-const SCHEMA_VERSION = 1;
+const SCHEMA_VERSION = 3;
-type MtimeRecord = { metaMs: number; vttMs: number | null };
+type Handling = "youtube" | "transcribe";
+type ChannelConfig = { handling: Handling; name?: string };
-type LiveSlug = {
- slug: string;
+// Primary sort/lookup key: [uploadDate, channelSlug, id]. Reverse iteration
+// yields newest-first directly (uploadDate is YYYYMMDD).
+type IndexKey = [string, string, string];
+// Path key: [channelSlug, videoDir]. Stable regardless of metadata contents,
+// so diffing by mtime doesn't require parsing metadata.info.json.
+type PathKey = [string, string];
+
+type MtimeRecord = {
+ metaMs: number;
+ transcriptMs: number | null;
+ indexKey: IndexKey;
+};
+
+type LiveEntry = {
+ channelSlug: string;
+ handling: Handling;
+ configName: string | undefined;
+ videoDir: string;
metaPath: string;
metaMs: number;
- vttPath: string;
- vttMs: number | null;
+ transcriptPath: string;
+ transcriptMs: number | null;
};
async function exists(p: string): Promise<boolean> {
@@ -44,33 +80,90 @@ async function exists(p: string): Promise<boolean> {
}
}
-async function scanSource(): Promise<LiveSlug[]> {
- const entries = await readdir(DATA_DIR, { withFileTypes: true });
- const out: LiveSlug[] = [];
- for (const e of entries) {
- if (!e.isDirectory()) continue;
- const slug = e.name;
- const metaPath = path.join(DATA_DIR, slug, "metadata.info.json");
- const vttPath = path.join(DATA_DIR, slug, "transcript.en.vtt");
- let metaMs: number;
- try {
- metaMs = (await stat(metaPath)).mtimeMs;
- } catch {
- // Skip slugs without metadata — they're not usable.
+async function readChannelConfig(dir: string): Promise<ChannelConfig | null> {
+ try {
+ const raw = await readFile(path.join(dir, "config.json"), "utf8");
+ const parsed = JSON.parse(raw) as ChannelConfig;
+ if (parsed.handling !== "youtube" && parsed.handling !== "transcribe") {
+ return null;
+ }
+ return parsed;
+ } catch {
+ return null;
+ }
+}
+
+async function scanSource(): Promise<{
+ live: LiveEntry[];
+ channels: Map<string, ChannelConfig>;
+}> {
+ const channels = new Map<string, ChannelConfig>();
+ const live: LiveEntry[] = [];
+ const channelEntries = await readdir(CHANNELS_DIR, { withFileTypes: true });
+ for (const ch of channelEntries) {
+ if (!ch.isDirectory()) continue;
+ const channelDir = path.join(CHANNELS_DIR, ch.name);
+ const cfg = await readChannelConfig(channelDir);
+ if (!cfg) {
+ console.warn(
+ `Skipping channel ${ch.name}: missing or invalid config.json`,
+ );
continue;
}
- let vttMs: number | null = null;
+ channels.set(ch.name, cfg);
+ const dataDir = path.join(channelDir, "data");
+ let videoEntries: Dirent[];
try {
- vttMs = (await stat(vttPath)).mtimeMs;
+ videoEntries = await readdir(dataDir, { withFileTypes: true });
} catch {
- vttMs = null;
+ continue;
+ }
+ const transcriptName =
+ cfg.handling === "youtube" ? "transcript.en.vtt" : "transcript.json";
+ for (const v of videoEntries) {
+ if (!v.isDirectory()) continue;
+ const videoDir = v.name;
+ const metaPath = path.join(dataDir, videoDir, "metadata.info.json");
+ const transcriptPath = path.join(dataDir, videoDir, transcriptName);
+ let metaMs: number;
+ try {
+ metaMs = (await stat(metaPath)).mtimeMs;
+ } catch {
+ continue;
+ }
+ let transcriptMs: number | null = null;
+ try {
+ transcriptMs = (await stat(transcriptPath)).mtimeMs;
+ } catch {
+ transcriptMs = null;
+ }
+ live.push({
+ channelSlug: ch.name,
+ handling: cfg.handling,
+ configName: cfg.name,
+ videoDir,
+ metaPath,
+ metaMs,
+ transcriptPath,
+ transcriptMs,
+ });
}
- out.push({ slug, metaPath, metaMs, vttPath, vttMs });
}
- return out;
+ return { live, channels };
}
-async function writeJsonAtomic(filePath: string, value: unknown): Promise<void> {
+function pathKeyId(k: PathKey): string {
+ return `${k[0]}\x00${k[1]}`;
+}
+
+function indexKeysEqual(a: IndexKey, b: IndexKey): boolean {
+ return a[0] === b[0] && a[1] === b[1] && a[2] === b[2];
+}
+
+async function writeJsonAtomic(
+ filePath: string,
+ value: unknown,
+): Promise<void> {
const tmp = `${filePath}.tmp-${process.pid}`;
await writeFile(tmp, JSON.stringify(value));
await rename(tmp, filePath);
@@ -87,15 +180,15 @@ async function main(): Promise<void> {
maxDbs: 8,
compression: true,
});
- const sums = root.openDB<TranscriptSummary, string>({
+ const sums = root.openDB<TranscriptSummary, IndexKey>({
name: "sums",
encoding: "msgpack",
});
- const cues = root.openDB<Cue[], string>({
+ const cues = root.openDB<Cue[], IndexKey>({
name: "cues",
encoding: "msgpack",
});
- const mtimes = root.openDB<MtimeRecord, string>({
+ const mtimes = root.openDB<MtimeRecord, PathKey>({
name: "mtimes",
encoding: "msgpack",
});
@@ -104,7 +197,6 @@ async function main(): Promise<void> {
encoding: "msgpack",
});
- // Force full rebuild if the schema changed.
const storedSchema = meta.get("schema") as number | undefined;
const schemaBumped = storedSchema !== SCHEMA_VERSION;
if (schemaBumped) {
@@ -117,32 +209,43 @@ async function main(): Promise<void> {
await meta.put("schema", SCHEMA_VERSION);
}
- const live = await scanSource();
- const liveBySlug = new Map(live.map((s) => [s.slug, s]));
-
- const known = new Set<string>();
- for (const { key } of mtimes.getRange()) known.add(key);
+ const { live, channels: channelConfigs } = await scanSource();
+ const livePathIds = new Set<string>();
+ const liveByPathId = new Map<string, LiveEntry>();
+ for (const s of live) {
+ const id = pathKeyId([s.channelSlug, s.videoDir]);
+ livePathIds.add(id);
+ liveByPathId.set(id, s);
+ }
- const added: LiveSlug[] = [];
- const changed: LiveSlug[] = [];
- const removed: string[] = [];
+ // Diff against prior mtimes (path-keyed — no metadata reads required).
+ const added: LiveEntry[] = [];
+ const changed: LiveEntry[] = [];
+ const removed: { pathKey: PathKey; indexKey: IndexKey }[] = [];
for (const s of live) {
- const prev = mtimes.get(s.slug);
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ const prev = mtimes.get(pk);
if (!prev) {
added.push(s);
- } else if (prev.metaMs !== s.metaMs || prev.vttMs !== s.vttMs) {
+ } else if (
+ prev.metaMs !== s.metaMs ||
+ prev.transcriptMs !== s.transcriptMs
+ ) {
changed.push(s);
}
}
- for (const slug of known) {
- if (!liveBySlug.has(slug)) removed.push(slug);
+ for (const { key, value } of mtimes.getRange()) {
+ const k = key as PathKey;
+ if (!livePathIds.has(pathKeyId(k))) {
+ removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey });
+ }
}
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
- // Short-circuit: if nothing changed AND public/ is intact, we're done.
+ // Short-circuit: nothing changed AND public/ is intact.
if (!anyMutations && !schemaBumped) {
const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null);
if (manifestRaw) {
@@ -153,7 +256,6 @@ async function main(): Promise<void> {
parsed.totalCount === live.length &&
parsed.pageSize === SUMMARIES_PAGE_SIZE
) {
- // Spot-check first and last page files exist.
const firstPage = path.join(SUMMARIES_DIR, pageFileName(0));
const lastPage = path.join(
SUMMARIES_DIR,
@@ -168,7 +270,7 @@ async function main(): Promise<void> {
}
}
} catch {
- // fall through to full emission
+ // fall through
}
}
}
@@ -177,7 +279,12 @@ async function main(): Promise<void> {
`Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`,
);
- // Process mutations: parse files and write to LMDB + per-slug public JSON.
+ // Ensure per-channel public/transcripts/<channelSlug>/ subdirs exist.
+ for (const channelSlug of channelConfigs.keys()) {
+ await mkdir(path.join(TRANSCRIPTS_DIR, channelSlug), { recursive: true });
+ }
+
+ // Process mutations.
const toProcess = [...added, ...changed];
let processed = 0;
const BATCH = 200;
@@ -188,28 +295,62 @@ async function main(): Promise<void> {
try {
const metaRaw = await readFile(s.metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
- const summary = summarize(s.slug, parsedMeta);
+ const summary = summarize(
+ s.channelSlug,
+ s.videoDir,
+ parsedMeta,
+ s.configName,
+ );
+ if (!summary.uploadDate) {
+ console.warn(
+ `Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`,
+ );
+ return;
+ }
+ const indexKey: IndexKey = [
+ summary.uploadDate,
+ s.channelSlug,
+ summary.id,
+ ];
let cueList: Cue[] | undefined;
- if (s.vttMs !== null) {
+ if (s.transcriptMs !== null) {
try {
- const vtt = await readFile(s.vttPath, "utf8");
- cueList = parseVtt(vtt);
+ const raw = await readFile(s.transcriptPath, "utf8");
+ cueList =
+ s.handling === "youtube" ? parseVtt(raw) : parseWhisper(raw);
} catch {
cueList = undefined;
}
}
- sums.put(s.slug, summary);
- if (cueList) cues.put(s.slug, cueList);
- else cues.remove(s.slug);
- mtimes.put(s.slug, { metaMs: s.metaMs, vttMs: s.vttMs });
+
+ // If upload_date shifted (metadata edit), the old composite key is
+ // stale — drop it before writing the new one.
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ const prev = mtimes.get(pk);
+ if (prev && !indexKeysEqual(prev.indexKey, indexKey)) {
+ sums.remove(prev.indexKey);
+ cues.remove(prev.indexKey);
+ }
+
+ sums.put(indexKey, summary);
+ if (cueList) cues.put(indexKey, cueList);
+ else cues.remove(indexKey);
+ mtimes.put(pk, {
+ metaMs: s.metaMs,
+ transcriptMs: s.transcriptMs,
+ indexKey,
+ });
const detail = { ...summary, cues: cueList };
await writeJsonAtomic(
- path.join(TRANSCRIPTS_DIR, `${s.slug}.json`),
+ path.join(TRANSCRIPTS_DIR, `${summary.slug}.json`),
detail,
);
} catch (err) {
- console.warn(`Failed to process ${s.slug}:`, err);
+ console.warn(
+ `Failed to process ${s.channelSlug}/${s.videoDir}:`,
+ err,
+ );
}
}),
);
@@ -221,22 +362,26 @@ async function main(): Promise<void> {
if (toProcess.length > BATCH) process.stdout.write("\n");
// Process removals.
- for (const slug of removed) {
- sums.remove(slug);
- cues.remove(slug);
- mtimes.remove(slug);
+ for (const { pathKey, indexKey } of removed) {
+ sums.remove(indexKey);
+ cues.remove(indexKey);
+ mtimes.remove(pathKey);
+ const slug = `${indexKey[1]}/${indexKey[2]}`;
await rm(path.join(TRANSCRIPTS_DIR, `${slug}.json`), { force: true });
}
- // Flush writes before we iterate for emission.
await sums.flushed;
await cues.flushed;
await mtimes.flushed;
- // Stream LMDB (reverse slug order = newest first) to emit paginated summaries.
- // Hold at most one page worth of summaries in memory at once.
+ // Stream LMDB (reverse composite-key order = newest-date first) to emit
+ // paginated summaries.
const pageSize = SUMMARIES_PAGE_SIZE;
- const channels = new Map<string, number>();
+ const channelCounts = new Map<string, number>();
+ // Seed channel list from config so empty channels still appear in the UI.
+ for (const cfg of channelConfigs.values()) {
+ if (cfg.name) channelCounts.set(cfg.name, 0);
+ }
let pageIndex = 0;
let buffer: DisplaySummary[] = [];
let total = 0;
@@ -249,18 +394,18 @@ async function main(): Promise<void> {
pageIndex++;
};
- // LMDB returns keys in ascending order; we want newest-first (descending slug).
- // Slugs are `YYYYMMDD_...` so lexicographic reverse == chronological reverse.
for (const { value } of sums.getRange({ reverse: true })) {
const s = value as TranscriptSummary;
- if (s.channel) channels.set(s.channel, (channels.get(s.channel) ?? 0) + 1);
+ if (s.channel) {
+ channelCounts.set(s.channel, (channelCounts.get(s.channel) ?? 0) + 1);
+ }
buffer.push(toDisplaySummary(s));
total++;
if (buffer.length >= pageSize) await flushPage();
}
await flushPage();
- // Prune stale page files (if page count shrank).
+ // Prune stale page files.
const expectedPages = new Set<string>();
for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i));
const existing = await readdir(SUMMARIES_DIR).catch(() => [] as string[]);
@@ -271,7 +416,7 @@ async function main(): Promise<void> {
}
}
- const channelList: ChannelEntry[] = Array.from(channels.entries())
+ const channelList: ChannelEntry[] = Array.from(channelCounts.entries())
.map(([name, count]) => ({ name, count }))
.sort((a, b) => a.name.localeCompare(b.name));