Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 4d5790520e64cd9c389bbc436d0e9156cdf68746
parent 66cfeb0a8097798cb8466116ea11caaa8ba2478b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 26 Apr 2026 20:51:59 -0400

refactor: extract index/whisper/verify into common/ controllers with CLI shims

Moves all data-mutating scripts into common/ as library functions and adds a
single getPaths() entry point so both apps and CLI shims agree on filesystem
locations.

- common/lib/paths.ts: getPaths() returns transcripts/jobs/lmdb/export paths
  and binary/model locations, all overridable via TRANSCRIPTS_DIR,
  EXPORT_PUBLIC_DIR, YTDLP_BIN, WHISPER_BIN, WHISPER_MODEL,
  PARALLEL_TRANSCRIBE_LIMIT
- common/lib/channelConfig.ts: typed ChannelConfig schema with optional url,
  audioFormat, cookiesFromBrowser, ytdlpExtraArgs, lastSyncedAt,
  lastFullDownloadAt fields (backward compatible with existing configs)
- common/lib/archive.ts: parses yt-dlp archive files into id sets
- common/controller/buildIndex.ts: body of the old build-index.ts with paths
  parameterized and console.log -> onLog. Tolerates a missing channels/ dir
  for fresh transcripts directories.
- common/controller/whisperBatch.ts: generalizes transform.ts +
  retry-failures.ts; supports modes "all" and "retry-failures", AbortSignal
  cancellation, atomic-rename of transcript.json to avoid half-written files.
- common/controller/verifyTranscripts.ts: generalizes verify-transcripts.ts
  with handling-aware transcript filename selection.
- common/controller/channels.ts: list/read/write/create/delete channels for
  the upcoming editor UI.
- common/bin/{build-index,transform,retry-failures,verify-transcripts}.ts:
  thin tsx shims that delegate to controllers; --channel <slug> arg.
- common/lib/settings.ts and transcripts.ts: route through getPaths().

- Deletes the in-tree transcripts/{transform,retry-failures,verify-transcripts,
  split-transcripts,rename-transcripts,rename-transcripts-old}.ts (one-offs
  preserved by the inner transcripts/.git).
- Lifts tsx, lmdb, execa, p-limit, fs-extra, @sindresorhus/slugify into
  common/package.json (dropped from export/).
- Updates export prebuild and root build:index to call the shim.

Verified: pnpm --filter export run build still produces 13,344 transcripts /
14 pages / 7 channels with byte-equivalent output. TRANSCRIPTS_DIR +
EXPORT_PUBLIC_DIR overrides isolate test runs from the real data.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/bin/_parseFlags.ts | 23+++++++++++++++++++++++
Acommon/bin/build-index.ts | 8++++++++
Acommon/bin/retry-failures.ts | 26++++++++++++++++++++++++++
Acommon/bin/transform.ts | 26++++++++++++++++++++++++++
Acommon/bin/verify-transcripts.ts | 23+++++++++++++++++++++++
Acommon/controller/buildIndex.ts | 674+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/channels.ts | 109+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/verifyTranscripts.ts | 87+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/whisperBatch.ts | 129+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/archive.ts | 40++++++++++++++++++++++++++++++++++++++++
Acommon/lib/channelConfig.ts | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/paths.ts | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/settings.ts | 14++------------
Mcommon/lib/transcripts.ts | 11+++--------
Mcommon/package.json | 5+++++
Mexport/package.json | 8+-------
Dexport/scripts/build-index.ts | 672-------------------------------------------------------------------------------
Mpackage.json | 2+-
Mpnpm-lock.yaml | 33+++++++++++++++------------------
19 files changed, 1293 insertions(+), 718 deletions(-)

diff --git a/common/bin/_parseFlags.ts b/common/bin/_parseFlags.ts @@ -0,0 +1,23 @@ +// Tiny argv parser: supports `--key value` and `--key=value`. Unknown +// positional args are dropped. Good enough for the bin shims; reach for +// commander/yargs if we ever need subcommands. +export function parseFlags(argv: string[]): Record<string, string> { + const out: Record<string, string> = {}; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (!a.startsWith("--")) continue; + const eq = a.indexOf("="); + if (eq > 0) { + out[a.slice(2, eq)] = a.slice(eq + 1); + } else { + const next = argv[i + 1]; + if (next && !next.startsWith("--")) { + out[a.slice(2)] = next; + i++; + } else { + out[a.slice(2)] = "true"; + } + } + } + return out; +} diff --git a/common/bin/build-index.ts b/common/bin/build-index.ts @@ -0,0 +1,8 @@ +#!/usr/bin/env tsx +import { getPaths } from "../lib/paths"; +import { buildIndex } from "../controller/buildIndex"; + +buildIndex({ paths: getPaths() }).catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/common/bin/retry-failures.ts b/common/bin/retry-failures.ts @@ -0,0 +1,26 @@ +#!/usr/bin/env tsx +import { getPaths } from "../lib/paths"; +import { runWhisperBatch } from "../controller/whisperBatch"; +import { parseFlags } from "./_parseFlags"; + +const flags = parseFlags(process.argv.slice(2)); +const channelSlug = flags.channel; +if (!channelSlug) { + console.error("Usage: retry-failures.ts --channel <slug>"); + process.exit(2); +} + +runWhisperBatch({ + channelSlug, + paths: getPaths(), + mode: "retry-failures", +}) + .then((result) => { + console.log( + `Done: ${result.succeeded} succeeded, ${result.failed} failed, ${result.skipped} skipped, ${result.attempted} attempted.`, + ); + }) + .catch((err) => { + console.error(err); + process.exit(1); + }); diff --git a/common/bin/transform.ts b/common/bin/transform.ts @@ -0,0 +1,26 @@ +#!/usr/bin/env tsx +import { getPaths } from "../lib/paths"; +import { runWhisperBatch } from "../controller/whisperBatch"; +import { parseFlags } from "./_parseFlags"; + +const flags = parseFlags(process.argv.slice(2)); +const channelSlug = flags.channel; +if (!channelSlug) { + console.error("Usage: transform.ts --channel <slug>"); + process.exit(2); +} + +runWhisperBatch({ + channelSlug, + paths: getPaths(), + mode: "all", +}) + .then((result) => { + console.log( + `Done: ${result.succeeded} succeeded, ${result.failed} failed, ${result.skipped} skipped, ${result.attempted} attempted.`, + ); + }) + .catch((err) => { + console.error(err); + process.exit(1); + }); diff --git a/common/bin/verify-transcripts.ts b/common/bin/verify-transcripts.ts @@ -0,0 +1,23 @@ +#!/usr/bin/env tsx +import { getPaths } from "../lib/paths"; +import { verifyTranscripts } from "../controller/verifyTranscripts"; +import { parseFlags } from "./_parseFlags"; + +const flags = parseFlags(process.argv.slice(2)); +const channelSlug = flags.channel; +if (!channelSlug) { + console.error("Usage: verify-transcripts.ts --channel <slug>"); + process.exit(2); +} + +verifyTranscripts({ channelSlug, paths: getPaths() }) + .then(({ duplicates, missing }) => { + console.log(`\n${duplicates.length} duplicate transcripts`); + if (duplicates.length > 0) console.log(duplicates.join(" ")); + console.log(`\n${missing.length} missing transcripts`); + if (missing.length > 0) console.log(missing.join(" ")); + }) + .catch((err) => { + console.error(err); + process.exit(1); + }); diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -0,0 +1,674 @@ +// Preprocess <transcriptsDir>/channels/<channelSlug>/data/<videoDir>/ into: +// - LMDB cache at <transcriptsDir>/index.mdb (for incremental rebuilds) +// - <exportSummariesDir>/{manifest,page-NNNN}.json (paginated cross-channel +// summaries index for browse/search) +// - <exportTranscriptsDir>/<channelSlug>/{manifest,page-NNNN}.json +// (per-channel paginated transcript detail + cues, oldest-first so +// adding a newer video only dirties the last page) +// +// Per-channel config.json selects the transcript parser ("youtube" → VTT, +// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match. + +import path from "node:path"; +import { createHash } from "node:crypto"; +import { + mkdir, + readdir, + readFile, + rename, + rm, + stat, + writeFile, +} from "node:fs/promises"; +import type { Dirent } from "node:fs"; +import { open } from "lmdb"; +import { parseVtt, type Cue } from "../lib/vtt"; +import { parseWhisper } from "../lib/whisper"; +import { + summarize, + toDisplaySummary, + type RawMetadata, +} from "../lib/transcripts-server"; +import type { + TranscriptSummary, + TranscriptDetail, + DisplaySummary, +} from "../lib/transcripts"; +import type { + Manifest, + ChannelEntry, + ChannelTranscriptsManifest, +} from "../lib/manifest"; +import { + MANIFEST_VERSION, + SUMMARIES_PAGE_SIZE, + TRANSCRIPTS_MANIFEST_VERSION, + pageFileName, + transcriptPageFileName, +} from "../lib/manifest"; +import { getSettings } from "../lib/settings"; +import { + parseChannelConfig, + type ChannelConfig, + type ChannelHandling, +} from "../lib/channelConfig"; +import type { Paths } from "../lib/paths"; + +const SCHEMA_VERSION = 5; + +type IndexKey = [string, string, string]; +type ChannelKey = [string, string, string]; +type PageHashKey = [string, number]; +type PathKey = [string, string]; + +type MtimeRecord = { + metaMs: number; + transcriptMs: number | null; + indexKey: IndexKey; +}; + +type PageHashRecord = { + hash: string; + entryCount: number; + sizeBytes: number; +}; + +function indexToChannelKey(k: IndexKey): ChannelKey { + return [k[1], k[0], k[2]]; +} + +type LiveEntry = { + channelSlug: string; + handling: ChannelHandling; + configName: string | undefined; + videoDir: string; + metaPath: string; + metaMs: number; + transcriptPath: string; + transcriptMs: number | null; +}; + +async function exists(p: string): Promise<boolean> { + try { + await stat(p); + return true; + } catch { + return false; + } +} + +async function readChannelConfigFile( + dir: string, +): Promise<ChannelConfig | null> { + try { + const raw = await readFile(path.join(dir, "config.json"), "utf8"); + return parseChannelConfig(JSON.parse(raw)); + } catch { + return null; + } +} + +async function scanSource( + channelsDir: string, + log: (msg: string) => void, +): Promise<{ + live: LiveEntry[]; + channels: Map<string, ChannelConfig>; +}> { + const channels = new Map<string, ChannelConfig>(); + const live: LiveEntry[] = []; + let channelEntries: Dirent[]; + try { + channelEntries = await readdir(channelsDir, { withFileTypes: true }); + } catch { + // Fresh transcripts dir with no channels yet. + return { live, channels }; + } + for (const ch of channelEntries) { + if (!ch.isDirectory()) continue; + const channelDir = path.join(channelsDir, ch.name); + const cfg = await readChannelConfigFile(channelDir); + if (!cfg) { + log(`Skipping channel ${ch.name}: missing or invalid config.json`); + continue; + } + channels.set(ch.name, cfg); + const dataDir = path.join(channelDir, "data"); + let videoEntries: Dirent[]; + try { + videoEntries = await readdir(dataDir, { withFileTypes: true }); + } catch { + continue; + } + const transcriptName = + cfg.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; + for (const v of videoEntries) { + if (!v.isDirectory()) continue; + const videoDir = v.name; + const metaPath = path.join(dataDir, videoDir, "metadata.info.json"); + const transcriptPath = path.join(dataDir, videoDir, transcriptName); + let metaMs: number; + try { + metaMs = (await stat(metaPath)).mtimeMs; + } catch { + continue; + } + let transcriptMs: number | null = null; + try { + transcriptMs = (await stat(transcriptPath)).mtimeMs; + } catch { + transcriptMs = null; + } + live.push({ + channelSlug: ch.name, + handling: cfg.handling, + configName: cfg.name, + videoDir, + metaPath, + metaMs, + transcriptPath, + transcriptMs, + }); + } + } + return { live, channels }; +} + +function pathKeyId(k: PathKey): string { + return `${k[0]}\x00${k[1]}`; +} + +function indexKeysEqual(a: IndexKey, b: IndexKey): boolean { + return a[0] === b[0] && a[1] === b[1] && a[2] === b[2]; +} + +async function writeJsonAtomic( + filePath: string, + value: unknown, +): Promise<void> { + const tmp = `${filePath}.tmp-${process.pid}`; + await writeFile(tmp, JSON.stringify(value)); + await rename(tmp, filePath); +} + +export type BuildIndexResult = { + totalCount: number; + pageCount: number; + channelCount: number; + durationMs: number; + pagesWritten: number; + pagesSkipped: number; + pagesDeleted: number; + added: number; + changed: number; + removed: number; + shortCircuited: boolean; +}; + +export type BuildIndexOptions = { + paths: Paths; + onLog?: (msg: string) => void; +}; + +export async function buildIndex({ + paths, + onLog, +}: BuildIndexOptions): Promise<BuildIndexResult> { + const log = onLog ?? ((msg: string) => console.log(msg)); + const t0 = Date.now(); + + const channelsDir = paths.channelsDir; + const dbPath = paths.lmdbPath; + const summariesDir = paths.exportSummariesDir; + const transcriptsOutDir = paths.exportTranscriptsDir; + const manifestPath = path.join(summariesDir, "manifest.json"); + + await mkdir(path.dirname(dbPath), { recursive: true }); + await mkdir(summariesDir, { recursive: true }); + await mkdir(transcriptsOutDir, { recursive: true }); + + const root = open({ + path: dbPath, + maxDbs: 8, + compression: true, + }); + const sums = root.openDB<TranscriptSummary, IndexKey>({ + name: "sums", + encoding: "msgpack", + }); + const cues = root.openDB<Cue[], IndexKey>({ + name: "cues", + encoding: "msgpack", + }); + const mtimes = root.openDB<MtimeRecord, PathKey>({ + name: "mtimes", + encoding: "msgpack", + }); + const byChannel = root.openDB<number, ChannelKey>({ + name: "byChannel", + encoding: "msgpack", + }); + const pageHashes = root.openDB<PageHashRecord, PageHashKey>({ + name: "pageHashes", + encoding: "msgpack", + }); + const meta = root.openDB<unknown, string>({ + name: "meta", + encoding: "msgpack", + }); + + const storedSchema = meta.get("schema") as number | undefined; + const schemaBumped = storedSchema !== SCHEMA_VERSION; + if (schemaBumped) { + log( + `Schema change (${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}); invalidating LMDB cache.`, + ); + await sums.clearAsync(); + await cues.clearAsync(); + await mtimes.clearAsync(); + await byChannel.clearAsync(); + await pageHashes.clearAsync(); + await meta.put("schema", SCHEMA_VERSION); + } + + const { live, channels: channelConfigs } = await scanSource(channelsDir, log); + const livePathIds = new Set<string>(); + const liveByPathId = new Map<string, LiveEntry>(); + for (const s of live) { + const id = pathKeyId([s.channelSlug, s.videoDir]); + livePathIds.add(id); + liveByPathId.set(id, s); + } + + const added: LiveEntry[] = []; + const changed: LiveEntry[] = []; + const removed: { pathKey: PathKey; indexKey: IndexKey }[] = []; + + for (const s of live) { + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); + if (!prev) { + added.push(s); + } else if ( + prev.metaMs !== s.metaMs || + prev.transcriptMs !== s.transcriptMs + ) { + changed.push(s); + } + } + for (const { key, value } of mtimes.getRange()) { + const k = key as PathKey; + if (!livePathIds.has(pathKeyId(k))) { + removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey }); + } + } + + const anyMutations = + added.length > 0 || changed.length > 0 || removed.length > 0; + + const { maxTranscriptPageBytes: configuredMaxPageBytes } = getSettings(); + if (!anyMutations && !schemaBumped) { + const manifestRaw = await readFile(manifestPath, "utf8").catch(() => null); + if (manifestRaw) { + try { + const parsed = JSON.parse(manifestRaw) as Manifest; + if ( + parsed.version === MANIFEST_VERSION && + parsed.totalCount === live.length && + parsed.pageSize === SUMMARIES_PAGE_SIZE + ) { + const firstPage = path.join(summariesDir, pageFileName(0)); + const lastPage = path.join( + summariesDir, + pageFileName(Math.max(0, parsed.pageCount - 1)), + ); + let transcriptsIntact = true; + for (const channelSlug of channelConfigs.keys()) { + const mPath = path.join( + transcriptsOutDir, + channelSlug, + "manifest.json", + ); + const raw = await readFile(mPath, "utf8").catch(() => null); + if (!raw) { + transcriptsIntact = false; + break; + } + try { + const cm = JSON.parse(raw) as ChannelTranscriptsManifest; + if ( + cm.version !== TRANSCRIPTS_MANIFEST_VERSION || + cm.maxPageBytes !== configuredMaxPageBytes + ) { + transcriptsIntact = false; + break; + } + } catch { + transcriptsIntact = false; + break; + } + } + if ( + transcriptsIntact && + (await exists(firstPage)) && + (await exists(lastPage)) + ) { + log( + `Index up to date (${live.length} transcripts, ${parsed.pageCount} pages). Skipping.`, + ); + await root.close(); + return { + totalCount: parsed.totalCount, + pageCount: parsed.pageCount, + channelCount: parsed.channels.length, + durationMs: Date.now() - t0, + pagesWritten: 0, + pagesSkipped: 0, + pagesDeleted: 0, + added: 0, + changed: 0, + removed: 0, + shortCircuited: true, + }; + } + } + } catch { + // fall through + } + } + } + + log( + `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, + ); + + for (const channelSlug of channelConfigs.keys()) { + await mkdir(path.join(transcriptsOutDir, channelSlug), { recursive: true }); + } + + const toProcess = [...added, ...changed]; + let processed = 0; + const BATCH = 200; + for (let i = 0; i < toProcess.length; i += BATCH) { + const slice = toProcess.slice(i, i + BATCH); + await Promise.all( + slice.map(async (s) => { + try { + const metaRaw = await readFile(s.metaPath, "utf8"); + const parsedMeta = JSON.parse(metaRaw) as RawMetadata; + const summary = summarize( + s.channelSlug, + s.videoDir, + parsedMeta, + s.configName, + ); + if (!summary.uploadDate) { + log(`Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`); + return; + } + const indexKey: IndexKey = [ + summary.uploadDate, + s.channelSlug, + summary.id, + ]; + let cueList: Cue[] | undefined; + if (s.transcriptMs !== null) { + try { + const raw = await readFile(s.transcriptPath, "utf8"); + cueList = + s.handling === "youtube" ? parseVtt(raw) : parseWhisper(raw); + } catch { + cueList = undefined; + } + } + + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); + if (prev && !indexKeysEqual(prev.indexKey, indexKey)) { + sums.remove(prev.indexKey); + cues.remove(prev.indexKey); + byChannel.remove(indexToChannelKey(prev.indexKey)); + } + + sums.put(indexKey, summary); + if (cueList) cues.put(indexKey, cueList); + else cues.remove(indexKey); + byChannel.put(indexToChannelKey(indexKey), 1); + mtimes.put(pk, { + metaMs: s.metaMs, + transcriptMs: s.transcriptMs, + indexKey, + }); + } catch (err) { + log(`Failed to process ${s.channelSlug}/${s.videoDir}: ${String(err)}`); + } + }), + ); + processed += slice.length; + if (toProcess.length > BATCH) { + log(` processed ${processed}/${toProcess.length}`); + } + } + + for (const { pathKey, indexKey } of removed) { + sums.remove(indexKey); + cues.remove(indexKey); + byChannel.remove(indexToChannelKey(indexKey)); + mtimes.remove(pathKey); + } + + await sums.flushed; + await cues.flushed; + await byChannel.flushed; + await mtimes.flushed; + + const maxTranscriptPageBytes = configuredMaxPageBytes; + const generatedAt = new Date().toISOString(); + let pagesWritten = 0; + let pagesSkipped = 0; + let pagesDeleted = 0; + + type PendingEntry = { encoded: string; id: string }; + + const writePage = async ( + channelSlug: string, + idx: number, + entries: PendingEntry[], + ): Promise<void> => { + const body = `[${entries.map((e) => e.encoded).join(",")}]`; + const hash = createHash("sha1").update(body).digest("hex"); + const prev = pageHashes.get([channelSlug, idx]); + const outPath = path.join( + transcriptsOutDir, + channelSlug, + transcriptPageFileName(idx), + ); + if (prev?.hash === hash && (await exists(outPath))) { + pagesSkipped++; + return; + } + const sizeBytes = Buffer.byteLength(body, "utf8"); + const tmp = `${outPath}.tmp-${process.pid}`; + await writeFile(tmp, body); + await rename(tmp, outPath); + pageHashes.put([channelSlug, idx], { + hash, + entryCount: entries.length, + sizeBytes, + }); + pagesWritten++; + }; + + for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + const channelDir = path.join(transcriptsOutDir, channelSlug); + await mkdir(channelDir, { recursive: true }); + + let pageIdx = 0; + let buffer: PendingEntry[] = []; + let payloadBytes = 0; + const slugToPage: Record<string, number> = {}; + + for (const { key } of byChannel.getRange({ + start: [channelSlug], + end: [channelSlug, "￿"], + })) { + const ck = key as ChannelKey; + if (ck[0] !== channelSlug) continue; + const indexKey: IndexKey = [ck[1], ck[0], ck[2]]; + const summary = sums.get(indexKey); + if (!summary) continue; + const cueList = cues.get(indexKey); + const detail: TranscriptDetail = { ...summary, cues: cueList }; + const encoded = JSON.stringify(detail); + const entryBytes = Buffer.byteLength(encoded, "utf8"); + const commaBytes = buffer.length === 0 ? 0 : 1; + const delta = entryBytes + commaBytes; + + if (buffer.length > 0 && payloadBytes + delta + 2 > maxTranscriptPageBytes) { + await writePage(channelSlug, pageIdx, buffer); + pageIdx++; + buffer = []; + payloadBytes = 0; + } + + if (entryBytes + 2 > maxTranscriptPageBytes) { + log( + ` ${summary.slug}: entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`, + ); + } + + slugToPage[summary.id] = pageIdx; + buffer.push({ encoded, id: summary.id }); + payloadBytes += delta; + } + + if (buffer.length > 0) { + await writePage(channelSlug, pageIdx, buffer); + pageIdx++; + } + + const pageCount = pageIdx; + + const keep = new Set<string>(["manifest.json"]); + for (let i = 0; i < pageCount; i++) keep.add(transcriptPageFileName(i)); + const existing = await readdir(channelDir).catch(() => [] as string[]); + for (const name of existing) { + if (keep.has(name)) continue; + await rm(path.join(channelDir, name), { force: true }); + pagesDeleted++; + } + for (const { key } of pageHashes.getRange({ + start: [channelSlug, pageCount], + end: [channelSlug, Number.MAX_SAFE_INTEGER], + })) { + pageHashes.remove(key as PageHashKey); + } + + const channelManifest: ChannelTranscriptsManifest = { + version: TRANSCRIPTS_MANIFEST_VERSION, + channelSlug, + pageCount, + maxPageBytes: maxTranscriptPageBytes, + generatedAt, + slugToPage, + }; + await writeJsonAtomic( + path.join(channelDir, "manifest.json"), + channelManifest, + ); + } + + await pageHashes.flushed; + + const topEntries = await readdir(transcriptsOutDir, { + withFileTypes: true, + }).catch(() => [] as Dirent[]); + for (const e of topEntries) { + if (e.isDirectory()) { + if (!channelConfigs.has(e.name)) { + await rm(path.join(transcriptsOutDir, e.name), { + recursive: true, + force: true, + }); + } + } else if (e.isFile()) { + await rm(path.join(transcriptsOutDir, e.name), { force: true }); + } + } + + log( + `Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`, + ); + + const pageSize = SUMMARIES_PAGE_SIZE; + const channelCounts = new Map<string, number>(); + for (const cfg of channelConfigs.values()) { + if (cfg.name) channelCounts.set(cfg.name, 0); + } + let pageIndex = 0; + let buffer: DisplaySummary[] = []; + let total = 0; + + const flushPage = async () => { + if (buffer.length === 0) return; + const p = path.join(summariesDir, pageFileName(pageIndex)); + await writeJsonAtomic(p, buffer); + buffer = []; + pageIndex++; + }; + + for (const { value } of sums.getRange({ reverse: true })) { + const s = value as TranscriptSummary; + if (s.channel) { + channelCounts.set(s.channel, (channelCounts.get(s.channel) ?? 0) + 1); + } + buffer.push(toDisplaySummary(s)); + total++; + if (buffer.length >= pageSize) await flushPage(); + } + await flushPage(); + + const expectedPages = new Set<string>(); + for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i)); + const existing = await readdir(summariesDir).catch(() => [] as string[]); + for (const name of existing) { + if (!name.startsWith("page-")) continue; + if (!expectedPages.has(name)) { + await rm(path.join(summariesDir, name), { force: true }); + } + } + + const channelList: ChannelEntry[] = Array.from(channelCounts.entries()) + .map(([name, count]) => ({ name, count })) + .sort((a, b) => a.name.localeCompare(b.name)); + + const manifest: Manifest = { + version: MANIFEST_VERSION, + totalCount: total, + pageSize, + pageCount: pageIndex, + generatedAt: new Date().toISOString(), + channels: channelList, + }; + await writeJsonAtomic(manifestPath, manifest); + + await root.close(); + + const durationMs = Date.now() - t0; + log( + `Done in ${(durationMs / 1000).toFixed(2)}s. ${total} transcripts across ${pageIndex} pages; ${channelList.length} channels.`, + ); + return { + totalCount: total, + pageCount: pageIndex, + channelCount: channelList.length, + durationMs, + pagesWritten, + pagesSkipped, + pagesDeleted, + added: added.length, + changed: changed.length, + removed: removed.length, + shortCircuited: false, + }; +} diff --git a/common/controller/channels.ts b/common/controller/channels.ts @@ -0,0 +1,109 @@ +import path from "node:path"; +import { readdir, readFile, writeFile, rename, rm, stat, mkdir } from "node:fs/promises"; +import type { Dirent } from "node:fs"; +import { parseChannelConfig, type ChannelConfig } from "../lib/channelConfig"; +import type { Paths } from "../lib/paths"; + +export type ChannelStat = { + slug: string; + config: ChannelConfig; + hasArchive: boolean; + hasPlaylist: boolean; + videoCount: number; +}; + +async function exists(p: string): Promise<boolean> { + try { + await stat(p); + return true; + } catch { + return false; + } +} + +export async function readChannelConfig( + paths: Paths, + slug: string, +): Promise<ChannelConfig | null> { + try { + const raw = await readFile( + path.join(paths.channelsDir, slug, "config.json"), + "utf8", + ); + return parseChannelConfig(JSON.parse(raw)); + } catch { + return null; + } +} + +export async function channelExists( + paths: Paths, + slug: string, +): Promise<boolean> { + return exists(path.join(paths.channelsDir, slug, "config.json")); +} + +export async function listChannels(paths: Paths): Promise<ChannelStat[]> { + let entries: Dirent[]; + try { + entries = await readdir(paths.channelsDir, { withFileTypes: true }); + } catch { + return []; + } + const out: ChannelStat[] = []; + for (const e of entries) { + if (!e.isDirectory()) continue; + const slug = e.name; + const config = await readChannelConfig(paths, slug); + if (!config) continue; + const channelDir = path.join(paths.channelsDir, slug); + const dataDir = path.join(channelDir, "data"); + let videoCount = 0; + try { + const dirs = await readdir(dataDir, { withFileTypes: true }); + videoCount = dirs.filter((d) => d.isDirectory()).length; + } catch { + videoCount = 0; + } + out.push({ + slug, + config, + hasArchive: await exists(path.join(channelDir, "archive")), + hasPlaylist: await exists(path.join(channelDir, "playlist")), + videoCount, + }); + } + return out.sort((a, b) => a.slug.localeCompare(b.slug)); +} + +export async function writeChannelConfig( + paths: Paths, + slug: string, + config: ChannelConfig, +): Promise<void> { + const dir = path.join(paths.channelsDir, slug); + await mkdir(dir, { recursive: true }); + const file = path.join(dir, "config.json"); + const tmp = `${file}.tmp-${process.pid}`; + await writeFile(tmp, JSON.stringify(config, null, 2) + "\n"); + await rename(tmp, file); +} + +export async function createChannel( + paths: Paths, + slug: string, + config: ChannelConfig, +): Promise<void> { + if (await channelExists(paths, slug)) { + throw new Error(`Channel ${slug} already exists`); + } + await writeChannelConfig(paths, slug, config); +} + +export async function deleteChannel( + paths: Paths, + slug: string, +): Promise<void> { + const dir = path.join(paths.channelsDir, slug); + await rm(dir, { recursive: true, force: true }); +} diff --git a/common/controller/verifyTranscripts.ts b/common/controller/verifyTranscripts.ts @@ -0,0 +1,87 @@ +import path from "node:path"; +import { readdir, readFile, stat } from "node:fs/promises"; +import type { Paths } from "../lib/paths"; +import { + parseChannelConfig, + type ChannelConfig, +} from "../lib/channelConfig"; +import { readArchive } from "../lib/archive"; + +export type VerifyTranscriptsOptions = { + channelSlug: string; + paths: Paths; +}; + +export type VerifyTranscriptsResult = { + duplicates: string[]; + missing: string[]; +}; + +async function readChannelConfig( + channelDir: string, +): Promise<ChannelConfig | null> { + try { + const raw = await readFile(path.join(channelDir, "config.json"), "utf8"); + return parseChannelConfig(JSON.parse(raw)); + } catch { + return null; + } +} + +async function exists(p: string): Promise<boolean> { + try { + await stat(p); + return true; + } catch { + return false; + } +} + +// For each ID in the archive, confirm there's a matching <id>/transcript.<ext> +// on disk. Reports missing transcripts and any duplicate video directories. +export async function verifyTranscripts({ + channelSlug, + paths, +}: VerifyTranscriptsOptions): Promise<VerifyTranscriptsResult> { + const channelDir = path.join(paths.channelsDir, channelSlug); + const dataDir = path.join(channelDir, "data"); + const archivePath = path.join(channelDir, "archive"); + + const config = await readChannelConfig(channelDir); + const transcriptFilename = + config?.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; + + const dirEntries = await readdir(dataDir).catch(() => [] as string[]); + + // Map ID -> directory name. The current layout is one dir per video where + // dirName === videoID; preserve detection of legacy "<date>_<id>-<title>" + // names by treating any duplicate as a duplicate. + const dirMap = new Map<string, string>(); + const duplicates = new Set<string>(); + for (const dir of dirEntries) { + const id = dir; + const existing = dirMap.get(id); + if (existing) { + duplicates.add(dir.length > existing.length ? dir : existing); + } + dirMap.set(id, dir); + } + + const archive = await readArchive(archivePath); + const missing = new Set<string>(); + for (const id of archive.ids) { + const dir = dirMap.get(id); + if (!dir) { + missing.add(id); + continue; + } + if (!(await exists(path.join(dataDir, dir, transcriptFilename)))) { + missing.add(id); + } + } + + return { + duplicates: Array.from(duplicates), + missing: Array.from(missing), + }; +} diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts @@ -0,0 +1,129 @@ +import path from "node:path"; +import fs from "fs-extra"; +import { execa } from "execa"; +import pLimit from "p-limit"; +import type { Paths } from "../lib/paths"; + +const { pathExists, readdir, appendFile, readFile, ensureFile, rename } = fs; + +export type WhisperBatchMode = "all" | "retry-failures"; + +export type WhisperBatchOptions = { + channelSlug: string; + paths: Paths; + mode: WhisperBatchMode; + concurrency?: number; + audioFilename?: string; + onLog?: (msg: string) => void; + signal?: AbortSignal; +}; + +export type WhisperBatchResult = { + attempted: number; + succeeded: number; + failed: number; + skipped: number; +}; + +export async function runWhisperBatch({ + channelSlug, + paths, + mode, + concurrency, + audioFilename = "audio.m4a", + onLog, + signal, +}: WhisperBatchOptions): Promise<WhisperBatchResult> { + const log = onLog ?? ((m: string) => console.log(m)); + const channelDir = path.join(paths.channelsDir, channelSlug); + const dataDir = path.join(channelDir, "data"); + const failureListFile = path.join(channelDir, "failed-transcriptions"); + const transcriptFilename = "transcript.json"; + const limit = pLimit(concurrency ?? paths.parallelTranscribeLimit); + + await ensureFile(failureListFile); + + let videoDirs: string[]; + if (mode === "retry-failures") { + const lines = await readFile(failureListFile, "utf-8"); + videoDirs = lines.split("\n").filter(Boolean); + } else { + videoDirs = await readdir(dataDir); + } + + const failedSet = new Set<string>( + (await readFile(failureListFile, "utf-8")).split("\n").filter(Boolean), + ); + + let attempted = 0; + let succeeded = 0; + let failed = 0; + let skipped = 0; + + await Promise.all( + videoDirs.map((videoDir) => + limit(async () => { + if (signal?.aborted) { + skipped++; + return; + } + const videoPath = path.join(dataDir, videoDir); + const transcriptFilePath = path.join(videoPath, transcriptFilename); + if (mode === "all" && failedSet.has(videoDir)) { + log(`Skipping previously failed transcription for ${videoDir}`); + skipped++; + return; + } + if (await pathExists(transcriptFilePath)) { + log(`Transcription for ${videoDir} already exists`); + skipped++; + return; + } + log(`Transcribe ${videoDir} start`); + const start = Date.now(); + attempted++; + try { + // Whisper writes "transcript.json" relative to cwd. Use a tmp name + // and rename so a SIGTERM mid-write can't leave a half-baked file. + const tmpBase = `transcript.tmp-${process.pid}`; + await execa( + paths.whisperBin, + [ + "-ojf", + "-l", + "en", + "-m", + paths.whisperModel, + "-of", + tmpBase, + audioFilename, + ], + { + cwd: videoPath, + cancelSignal: signal, + }, + ); + const tmpPath = path.join(videoPath, `${tmpBase}.json`); + await rename(tmpPath, transcriptFilePath); + log( + `Transcribe ${videoDir} done in ${((Date.now() - start) / 1000).toFixed(2)}s`, + ); + succeeded++; + } catch (err) { + if (signal?.aborted) { + log(`Transcribe ${videoDir} cancelled`); + skipped++; + return; + } + log(`FAILED TO TRANSCRIBE ${videoDir}: ${String(err)}`); + if (mode === "all") { + await appendFile(failureListFile, `${videoDir}\n`); + } + failed++; + } + }), + ), + ); + + return { attempted, succeeded, failed, skipped }; +} diff --git a/common/lib/archive.ts b/common/lib/archive.ts @@ -0,0 +1,40 @@ +import { readFile } from "node:fs/promises"; + +export type ArchiveContents = { + ids: Set<string>; + byExtractor: Map<string, Set<string>>; +}; + +// yt-dlp archive lines have the form `<extractor> <id>` (e.g., `youtube abc123`, +// `lbry 9720133ae3a206487321f21eb5bdcc29f40828fc`). +export function parseArchive(text: string): ArchiveContents { + const ids = new Set<string>(); + const byExtractor = new Map<string, Set<string>>(); + for (const rawLine of text.split("\n")) { + const line = rawLine.trim(); + if (!line) continue; + const spaceIdx = line.indexOf(" "); + if (spaceIdx <= 0) continue; + const extractor = line.slice(0, spaceIdx); + const id = line.slice(spaceIdx + 1).trim(); + if (!id) continue; + ids.add(id); + let set = byExtractor.get(extractor); + if (!set) { + set = new Set<string>(); + byExtractor.set(extractor, set); + } + set.add(id); + } + return { ids, byExtractor }; +} + +export async function readArchive(filePath: string): Promise<ArchiveContents> { + let text: string; + try { + text = await readFile(filePath, "utf8"); + } catch { + return { ids: new Set(), byExtractor: new Map() }; + } + return parseArchive(text); +} diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -0,0 +1,53 @@ +export type ChannelHandling = "youtube" | "transcribe"; + +export type ChannelConfig = { + handling: ChannelHandling; + name?: string; + url?: string; + audioFormat?: "m4a" | "mp3" | "opus"; + cookiesFromBrowser?: string; + ytdlpExtraArgs?: string[]; + lastSyncedAt?: string; + lastFullDownloadAt?: string; +}; + +export const HANDLING_VALUES: ReadonlyArray<ChannelHandling> = [ + "youtube", + "transcribe", +]; + +export const AUDIO_FORMAT_VALUES: ReadonlyArray<NonNullable<ChannelConfig["audioFormat"]>> = [ + "m4a", + "mp3", + "opus", +]; + +export function parseChannelConfig(raw: unknown): ChannelConfig | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + if (r.handling !== "youtube" && r.handling !== "transcribe") return null; + const config: ChannelConfig = { handling: r.handling }; + if (typeof r.name === "string") config.name = r.name; + if (typeof r.url === "string") config.url = r.url; + if ( + r.audioFormat === "m4a" || + r.audioFormat === "mp3" || + r.audioFormat === "opus" + ) { + config.audioFormat = r.audioFormat; + } + if (typeof r.cookiesFromBrowser === "string") { + config.cookiesFromBrowser = r.cookiesFromBrowser; + } + if ( + Array.isArray(r.ytdlpExtraArgs) && + r.ytdlpExtraArgs.every((x) => typeof x === "string") + ) { + config.ytdlpExtraArgs = r.ytdlpExtraArgs as string[]; + } + if (typeof r.lastSyncedAt === "string") config.lastSyncedAt = r.lastSyncedAt; + if (typeof r.lastFullDownloadAt === "string") { + config.lastFullDownloadAt = r.lastFullDownloadAt; + } + return config; +} diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -0,0 +1,68 @@ +import fs from "node:fs"; +import path from "node:path"; +import os from "node:os"; + +export type Paths = { + monorepoRoot: string; + transcriptsDir: string; + channelsDir: string; + jobsDir: string; + lmdbPath: string; + exportDir: string; + exportPublicDir: string; + exportSummariesDir: string; + exportTranscriptsDir: string; + ytdlpBin: string; + whisperBin: string; + whisperModel: string; + parallelTranscribeLimit: number; +}; + +let cached: Paths | null = null; + +export function getPaths(): Paths { + if (cached) return cached; + const monorepoRoot = findMonorepoRoot(); + const transcriptsDir = + process.env.TRANSCRIPTS_DIR ?? path.join(monorepoRoot, "transcripts"); + const exportDir = path.join(monorepoRoot, "export"); + const exportPublicDir = + process.env.EXPORT_PUBLIC_DIR ?? path.join(exportDir, "public"); + cached = { + monorepoRoot, + transcriptsDir, + channelsDir: path.join(transcriptsDir, "channels"), + jobsDir: path.join(transcriptsDir, ".jobs"), + lmdbPath: path.join(transcriptsDir, "index.mdb"), + exportDir, + exportPublicDir, + exportSummariesDir: path.join(exportPublicDir, "summaries"), + exportTranscriptsDir: path.join(exportPublicDir, "transcripts"), + ytdlpBin: process.env.YTDLP_BIN ?? "yt-dlp", + whisperBin: process.env.WHISPER_BIN ?? "whisper-cli", + whisperModel: + process.env.WHISPER_MODEL ?? + path.join( + os.homedir(), + "whispercpp", + "whisper.cpp", + "models", + "ggml-base.en.bin", + ), + parallelTranscribeLimit: process.env.PARALLEL_TRANSCRIBE_LIMIT + ? Number(process.env.PARALLEL_TRANSCRIBE_LIMIT) + : 4, + }; + return cached; +} + +function findMonorepoRoot(): string { + let dir = process.cwd(); + for (let i = 0; i < 8; i++) { + if (fs.existsSync(path.join(dir, "pnpm-workspace.yaml"))) return dir; + const parent = path.dirname(dir); + if (parent === dir) break; + dir = parent; + } + return process.cwd(); +} diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -1,16 +1,6 @@ import fs from "node:fs"; import path from "node:path"; - -function findMonorepoRoot(): string { - let dir = process.cwd(); - for (let i = 0; i < 8; i++) { - if (fs.existsSync(path.join(dir, "pnpm-workspace.yaml"))) return dir; - const parent = path.dirname(dir); - if (parent === dir) break; - dir = parent; - } - return process.cwd(); -} +import { getPaths } from "./paths"; export type SiteSettings = { siteTitle: string; @@ -36,7 +26,7 @@ let cached: SiteSettings | null = null; export function getSettings(): SiteSettings { if (cached) return cached; - const file = path.join(findMonorepoRoot(), "settings.json"); + const file = path.join(getPaths().monorepoRoot, "settings.json"); let parsed: Partial<SiteSettings> = {}; try { parsed = JSON.parse(fs.readFileSync(file, "utf8")) as Partial<SiteSettings>; diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts @@ -1,13 +1,7 @@ import path from "node:path"; import type { Cue } from "./vtt"; import { readFile } from "fs-extra"; - -const MANIFEST_PATH = path.join( - process.cwd(), - "public", - "summaries", - "manifest.json", -); +import { getPaths } from "./paths"; export type Platform = "youtube" | "rumble" | "odysee"; @@ -48,7 +42,8 @@ export type TranscriptDetail = TranscriptSummary & { export type TranscriptPage = TranscriptDetail[]; export async function countTranscripts(): Promise<number> { - const raw = await readFile(MANIFEST_PATH, "utf8"); + const manifestPath = path.join(getPaths().exportSummariesDir, "manifest.json"); + const raw = await readFile(manifestPath, "utf8"); const parsed = JSON.parse(raw) as { totalCount?: number }; if (typeof parsed.totalCount !== "number") { throw new Error("manifest.json missing totalCount; run `pnpm build:index`"); diff --git a/common/package.json b/common/package.json @@ -4,8 +4,12 @@ "private": true, "type": "module", "dependencies": { + "@sindresorhus/slugify": "^3.0.0", "@tanstack/react-query": "^5.99.1", + "execa": "^9.6.1", "fs-extra": "^11.3.4", + "lmdb": "^3.5.4", + "p-limit": "^7.3.0", "react-player": "^2.16.1" }, "peerDependencies": { @@ -18,6 +22,7 @@ "@types/node": "^20.19.39", "@types/react": "^19.2.14", "@types/react-dom": "^19.2.3", + "tsx": "^4.21.0", "typescript": "^5.9.3" } } diff --git a/export/package.json b/export/package.json @@ -5,19 +5,15 @@ "type": "module", "scripts": { "dev": "next dev", - "build:index": "tsx scripts/build-index.ts", + "build:index": "tsx ../common/bin/build-index.ts", "prebuild": "pnpm run build:index", "build": "next build", "start": "serve out", "lint": "eslint" }, "dependencies": { - "@sindresorhus/slugify": "^3.0.0", "@tanstack/react-query": "^5.99.1", - "execa": "^9.6.1", - "lmdb": "^3.5.4", "next": "16.2.3", - "p-limit": "^7.3.0", "react": "19.2.4", "react-dom": "19.2.4", "react-player": "^2.16.1", @@ -26,13 +22,11 @@ }, "devDependencies": { "@tailwindcss/postcss": "^4.2.2", - "@types/fs-extra": "^11.0.4", "@types/node": "^20.19.39", "@types/react": "^19.2.14", "@types/react-dom": "^19.2.3", "eslint": "^9.39.4", "eslint-config-next": "16.2.3", - "fs-extra": "^11.3.4", "tailwindcss": "^4.2.2", "tsx": "^4.21.0", "typescript": "^5.9.3" diff --git a/export/scripts/build-index.ts b/export/scripts/build-index.ts @@ -1,672 +0,0 @@ -#!/usr/bin/env tsx -// Preprocess transcripts/channels/<channelSlug>/data/<videoDir>/ into: -// - LMDB cache at transcripts/index.mdb (for incremental rebuilds) -// - public/summaries/{manifest,page-NNNN}.json (paginated cross-channel -// summaries index for browse/search) -// - public/transcripts/<channelSlug>/{manifest,page-NNNN}.json -// (per-channel paginated transcript detail + cues, oldest-first so -// adding a newer video only dirties the last page) -// -// Per-channel config.json selects the transcript parser ("youtube" → VTT, -// "transcribe" → whisper.cpp JSON). Short-circuits when mtimes already match. - -import path from "node:path"; -import { createHash } from "node:crypto"; -import { - mkdir, - readdir, - readFile, - rename, - rm, - stat, - writeFile, -} from "node:fs/promises"; -import type { Dirent } from "node:fs"; -import { open } from "lmdb"; -import { fileURLToPath } from "node:url"; -import { parseVtt, type Cue } from "yt-dlp-transcript-common/lib/vtt"; -import { parseWhisper } from "yt-dlp-transcript-common/lib/whisper"; -import { - summarize, - toDisplaySummary, - type RawMetadata, -} from "yt-dlp-transcript-common/lib/transcripts-server"; -import type { - TranscriptSummary, - TranscriptDetail, - DisplaySummary, -} from "yt-dlp-transcript-common/lib/transcripts"; -import type { - Manifest, - ChannelEntry, - ChannelTranscriptsManifest, -} from "yt-dlp-transcript-common/lib/manifest"; -import { - MANIFEST_VERSION, - SUMMARIES_PAGE_SIZE, - TRANSCRIPTS_MANIFEST_VERSION, - pageFileName, - transcriptPageFileName, -} from "yt-dlp-transcript-common/lib/manifest"; -import { getSettings } from "yt-dlp-transcript-common/lib/settings"; - -// Self-locating: the script lives at <repoRoot>/export/scripts/build-index.ts, -// so the repo root is two levels up regardless of the cwd it was invoked from. -const __dirname = path.dirname(fileURLToPath(import.meta.url)); -const ROOT = path.resolve(__dirname, "..", ".."); -const CHANNELS_DIR = path.join(ROOT, "transcripts", "channels"); -const DB_PATH = path.join(ROOT, "transcripts", "index.mdb"); -const PUBLIC_DIR = path.join(ROOT, "export", "public"); -const SUMMARIES_DIR = path.join(PUBLIC_DIR, "summaries"); -const TRANSCRIPTS_DIR = path.join(PUBLIC_DIR, "transcripts"); -const MANIFEST_PATH = path.join(SUMMARIES_DIR, "manifest.json"); - -const SCHEMA_VERSION = 5; - -type Handling = "youtube" | "transcribe"; -type ChannelConfig = { handling: Handling; name?: string }; - -// Primary sort/lookup key: [uploadDate, channelSlug, id]. Reverse iteration -// yields newest-first directly (uploadDate is YYYYMMDD). -type IndexKey = [string, string, string]; -// Per-channel forward-scan index: [channelSlug, uploadDate, id]. -type ChannelKey = [string, string, string]; -// Page-hash key: [channelSlug, pageIndex]. -type PageHashKey = [string, number]; -// Path key: [channelSlug, videoDir]. Stable regardless of metadata contents, -// so diffing by mtime doesn't require parsing metadata.info.json. -type PathKey = [string, string]; - -type MtimeRecord = { - metaMs: number; - transcriptMs: number | null; - indexKey: IndexKey; -}; - -type PageHashRecord = { - hash: string; - entryCount: number; - sizeBytes: number; -}; - -function indexToChannelKey(k: IndexKey): ChannelKey { - return [k[1], k[0], k[2]]; -} - -type LiveEntry = { - channelSlug: string; - handling: Handling; - configName: string | undefined; - videoDir: string; - metaPath: string; - metaMs: number; - transcriptPath: string; - transcriptMs: number | null; -}; - -async function exists(p: string): Promise<boolean> { - try { - await stat(p); - return true; - } catch { - return false; - } -} - -async function readChannelConfig(dir: string): Promise<ChannelConfig | null> { - try { - const raw = await readFile(path.join(dir, "config.json"), "utf8"); - const parsed = JSON.parse(raw) as ChannelConfig; - if (parsed.handling !== "youtube" && parsed.handling !== "transcribe") { - return null; - } - return parsed; - } catch { - return null; - } -} - -async function scanSource(): Promise<{ - live: LiveEntry[]; - channels: Map<string, ChannelConfig>; -}> { - const channels = new Map<string, ChannelConfig>(); - const live: LiveEntry[] = []; - const channelEntries = await readdir(CHANNELS_DIR, { withFileTypes: true }); - for (const ch of channelEntries) { - if (!ch.isDirectory()) continue; - const channelDir = path.join(CHANNELS_DIR, ch.name); - const cfg = await readChannelConfig(channelDir); - if (!cfg) { - console.warn( - `Skipping channel ${ch.name}: missing or invalid config.json`, - ); - continue; - } - channels.set(ch.name, cfg); - const dataDir = path.join(channelDir, "data"); - let videoEntries: Dirent[]; - try { - videoEntries = await readdir(dataDir, { withFileTypes: true }); - } catch { - continue; - } - const transcriptName = - cfg.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; - for (const v of videoEntries) { - if (!v.isDirectory()) continue; - const videoDir = v.name; - const metaPath = path.join(dataDir, videoDir, "metadata.info.json"); - const transcriptPath = path.join(dataDir, videoDir, transcriptName); - let metaMs: number; - try { - metaMs = (await stat(metaPath)).mtimeMs; - } catch { - continue; - } - let transcriptMs: number | null = null; - try { - transcriptMs = (await stat(transcriptPath)).mtimeMs; - } catch { - transcriptMs = null; - } - live.push({ - channelSlug: ch.name, - handling: cfg.handling, - configName: cfg.name, - videoDir, - metaPath, - metaMs, - transcriptPath, - transcriptMs, - }); - } - } - return { live, channels }; -} - -function pathKeyId(k: PathKey): string { - return `${k[0]}\x00${k[1]}`; -} - -function indexKeysEqual(a: IndexKey, b: IndexKey): boolean { - return a[0] === b[0] && a[1] === b[1] && a[2] === b[2]; -} - -async function writeJsonAtomic( - filePath: string, - value: unknown, -): Promise<void> { - const tmp = `${filePath}.tmp-${process.pid}`; - await writeFile(tmp, JSON.stringify(value)); - await rename(tmp, filePath); -} - -async function main(): Promise<void> { - const t0 = Date.now(); - await mkdir(path.dirname(DB_PATH), { recursive: true }); - await mkdir(SUMMARIES_DIR, { recursive: true }); - await mkdir(TRANSCRIPTS_DIR, { recursive: true }); - - const root = open({ - path: DB_PATH, - maxDbs: 8, - compression: true, - }); - const sums = root.openDB<TranscriptSummary, IndexKey>({ - name: "sums", - encoding: "msgpack", - }); - const cues = root.openDB<Cue[], IndexKey>({ - name: "cues", - encoding: "msgpack", - }); - const mtimes = root.openDB<MtimeRecord, PathKey>({ - name: "mtimes", - encoding: "msgpack", - }); - // Per-channel forward index: lets us stream [channelSlug, uploadDate, id] - // in oldest-first order for size-based page packing without loading every - // channel's entries into memory. - const byChannel = root.openDB<number, ChannelKey>({ - name: "byChannel", - encoding: "msgpack", - }); - // Content hashes of emitted transcript pages, so unchanged pages skip the - // write. Since page 1 = oldest videos, adding a new newest-upload video - // only dirties the last page's hash. - const pageHashes = root.openDB<PageHashRecord, PageHashKey>({ - name: "pageHashes", - encoding: "msgpack", - }); - const meta = root.openDB<unknown, string>({ - name: "meta", - encoding: "msgpack", - }); - - const storedSchema = meta.get("schema") as number | undefined; - const schemaBumped = storedSchema !== SCHEMA_VERSION; - if (schemaBumped) { - console.log( - `Schema change (${storedSchema ?? "<none>"} -> ${SCHEMA_VERSION}); invalidating LMDB cache.`, - ); - await sums.clearAsync(); - await cues.clearAsync(); - await mtimes.clearAsync(); - await byChannel.clearAsync(); - await pageHashes.clearAsync(); - await meta.put("schema", SCHEMA_VERSION); - } - - const { live, channels: channelConfigs } = await scanSource(); - const livePathIds = new Set<string>(); - const liveByPathId = new Map<string, LiveEntry>(); - for (const s of live) { - const id = pathKeyId([s.channelSlug, s.videoDir]); - livePathIds.add(id); - liveByPathId.set(id, s); - } - - // Diff against prior mtimes (path-keyed — no metadata reads required). - const added: LiveEntry[] = []; - const changed: LiveEntry[] = []; - const removed: { pathKey: PathKey; indexKey: IndexKey }[] = []; - - for (const s of live) { - const pk: PathKey = [s.channelSlug, s.videoDir]; - const prev = mtimes.get(pk); - if (!prev) { - added.push(s); - } else if ( - prev.metaMs !== s.metaMs || - prev.transcriptMs !== s.transcriptMs - ) { - changed.push(s); - } - } - for (const { key, value } of mtimes.getRange()) { - const k = key as PathKey; - if (!livePathIds.has(pathKeyId(k))) { - removed.push({ pathKey: k, indexKey: (value as MtimeRecord).indexKey }); - } - } - - const anyMutations = - added.length > 0 || changed.length > 0 || removed.length > 0; - - // Short-circuit: nothing changed AND public/ is intact AND the configured - // max page size matches what each channel was last paginated with. - const { maxTranscriptPageBytes: configuredMaxPageBytes } = getSettings(); - if (!anyMutations && !schemaBumped) { - const manifestRaw = await readFile(MANIFEST_PATH, "utf8").catch(() => null); - if (manifestRaw) { - try { - const parsed = JSON.parse(manifestRaw) as Manifest; - if ( - parsed.version === MANIFEST_VERSION && - parsed.totalCount === live.length && - parsed.pageSize === SUMMARIES_PAGE_SIZE - ) { - const firstPage = path.join(SUMMARIES_DIR, pageFileName(0)); - const lastPage = path.join( - SUMMARIES_DIR, - pageFileName(Math.max(0, parsed.pageCount - 1)), - ); - let transcriptsIntact = true; - for (const channelSlug of channelConfigs.keys()) { - const mPath = path.join( - TRANSCRIPTS_DIR, - channelSlug, - "manifest.json", - ); - const raw = await readFile(mPath, "utf8").catch(() => null); - if (!raw) { - transcriptsIntact = false; - break; - } - try { - const cm = JSON.parse(raw) as ChannelTranscriptsManifest; - if ( - cm.version !== TRANSCRIPTS_MANIFEST_VERSION || - cm.maxPageBytes !== configuredMaxPageBytes - ) { - transcriptsIntact = false; - break; - } - } catch { - transcriptsIntact = false; - break; - } - } - if ( - transcriptsIntact && - (await exists(firstPage)) && - (await exists(lastPage)) - ) { - console.log( - `Index up to date (${live.length} transcripts, ${parsed.pageCount} pages). Skipping.`, - ); - await root.close(); - return; - } - } - } catch { - // fall through - } - } - } - - console.log( - `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, - ); - - // Ensure per-channel public/transcripts/<channelSlug>/ subdirs exist. - for (const channelSlug of channelConfigs.keys()) { - await mkdir(path.join(TRANSCRIPTS_DIR, channelSlug), { recursive: true }); - } - - // Process mutations. - const toProcess = [...added, ...changed]; - let processed = 0; - const BATCH = 200; - for (let i = 0; i < toProcess.length; i += BATCH) { - const slice = toProcess.slice(i, i + BATCH); - await Promise.all( - slice.map(async (s) => { - try { - const metaRaw = await readFile(s.metaPath, "utf8"); - const parsedMeta = JSON.parse(metaRaw) as RawMetadata; - const summary = summarize( - s.channelSlug, - s.videoDir, - parsedMeta, - s.configName, - ); - if (!summary.uploadDate) { - console.warn( - `Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`, - ); - return; - } - const indexKey: IndexKey = [ - summary.uploadDate, - s.channelSlug, - summary.id, - ]; - let cueList: Cue[] | undefined; - if (s.transcriptMs !== null) { - try { - const raw = await readFile(s.transcriptPath, "utf8"); - cueList = - s.handling === "youtube" ? parseVtt(raw) : parseWhisper(raw); - } catch { - cueList = undefined; - } - } - - // If upload_date shifted (metadata edit), the old composite key is - // stale — drop it before writing the new one. - const pk: PathKey = [s.channelSlug, s.videoDir]; - const prev = mtimes.get(pk); - if (prev && !indexKeysEqual(prev.indexKey, indexKey)) { - sums.remove(prev.indexKey); - cues.remove(prev.indexKey); - byChannel.remove(indexToChannelKey(prev.indexKey)); - } - - sums.put(indexKey, summary); - if (cueList) cues.put(indexKey, cueList); - else cues.remove(indexKey); - byChannel.put(indexToChannelKey(indexKey), 1); - mtimes.put(pk, { - metaMs: s.metaMs, - transcriptMs: s.transcriptMs, - indexKey, - }); - } catch (err) { - console.warn( - `Failed to process ${s.channelSlug}/${s.videoDir}:`, - err, - ); - } - }), - ); - processed += slice.length; - if (toProcess.length > BATCH) { - process.stdout.write(` processed ${processed}/${toProcess.length}\r`); - } - } - if (toProcess.length > BATCH) process.stdout.write("\n"); - - // Process removals. - for (const { pathKey, indexKey } of removed) { - sums.remove(indexKey); - cues.remove(indexKey); - byChannel.remove(indexToChannelKey(indexKey)); - mtimes.remove(pathKey); - } - - await sums.flushed; - await cues.flushed; - await byChannel.flushed; - await mtimes.flushed; - - // Emit per-channel paginated transcript pages. Pages are size-packed - // oldest-first, so newer videos land on the last page and older pages - // hash-match from build to build. - const maxTranscriptPageBytes = configuredMaxPageBytes; - const generatedAt = new Date().toISOString(); - let pagesWritten = 0; - let pagesSkipped = 0; - let pagesDeleted = 0; - - type PendingEntry = { encoded: string; id: string }; - - const writePage = async ( - channelSlug: string, - idx: number, - entries: PendingEntry[], - ): Promise<void> => { - const body = `[${entries.map((e) => e.encoded).join(",")}]`; - const hash = createHash("sha1").update(body).digest("hex"); - const prev = pageHashes.get([channelSlug, idx]); - const outPath = path.join( - TRANSCRIPTS_DIR, - channelSlug, - transcriptPageFileName(idx), - ); - if (prev?.hash === hash && (await exists(outPath))) { - pagesSkipped++; - return; - } - const sizeBytes = Buffer.byteLength(body, "utf8"); - const tmp = `${outPath}.tmp-${process.pid}`; - await writeFile(tmp, body); - await rename(tmp, outPath); - pageHashes.put([channelSlug, idx], { - hash, - entryCount: entries.length, - sizeBytes, - }); - pagesWritten++; - }; - - for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { - const channelDir = path.join(TRANSCRIPTS_DIR, channelSlug); - await mkdir(channelDir, { recursive: true }); - - let pageIdx = 0; - let buffer: PendingEntry[] = []; - // Running payload size excluding the outer `[` and `]` (accounted at - // flush time). Includes the leading commas between elements. - let payloadBytes = 0; - const slugToPage: Record<string, number> = {}; - - for (const { key } of byChannel.getRange({ - start: [channelSlug], - end: [channelSlug, "￿"], - })) { - const ck = key as ChannelKey; - if (ck[0] !== channelSlug) continue; - const indexKey: IndexKey = [ck[1], ck[0], ck[2]]; - const summary = sums.get(indexKey); - if (!summary) continue; - const cueList = cues.get(indexKey); - const detail: TranscriptDetail = { ...summary, cues: cueList }; - const encoded = JSON.stringify(detail); - const entryBytes = Buffer.byteLength(encoded, "utf8"); - const commaBytes = buffer.length === 0 ? 0 : 1; - const delta = entryBytes + commaBytes; - - // Close current page if this entry would push it past the limit. - // `+ 2` accounts for the outer brackets. - if (buffer.length > 0 && payloadBytes + delta + 2 > maxTranscriptPageBytes) { - await writePage(channelSlug, pageIdx, buffer); - pageIdx++; - buffer = []; - payloadBytes = 0; - } - - if (entryBytes + 2 > maxTranscriptPageBytes) { - console.warn( - ` ${summary.slug}: entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`, - ); - } - - slugToPage[summary.id] = pageIdx; - buffer.push({ encoded, id: summary.id }); - payloadBytes += delta; - } - - if (buffer.length > 0) { - await writePage(channelSlug, pageIdx, buffer); - pageIdx++; - } - - const pageCount = pageIdx; - - // Prune stale per-channel page files + drop pageHashes beyond pageCount. - const keep = new Set<string>(["manifest.json"]); - for (let i = 0; i < pageCount; i++) keep.add(transcriptPageFileName(i)); - const existing = await readdir(channelDir).catch(() => [] as string[]); - for (const name of existing) { - if (keep.has(name)) continue; - await rm(path.join(channelDir, name), { force: true }); - pagesDeleted++; - } - // pageHashes may still have records for page indexes >= pageCount - // (channel shrank). Drop them. - for (const { key } of pageHashes.getRange({ - start: [channelSlug, pageCount], - end: [channelSlug, Number.MAX_SAFE_INTEGER], - })) { - pageHashes.remove(key as PageHashKey); - } - - const channelManifest: ChannelTranscriptsManifest = { - version: TRANSCRIPTS_MANIFEST_VERSION, - channelSlug, - pageCount, - maxPageBytes: maxTranscriptPageBytes, - generatedAt, - slugToPage, - }; - await writeJsonAtomic( - path.join(channelDir, "manifest.json"), - channelManifest, - ); - } - - await pageHashes.flushed; - - // Prune channel directories for channels no longer in config, and any - // legacy per-video `<id>.json` files that leaked in from pre-schema-4 - // builds at the top level of transcripts/ (pre-channel-subdir layout). - const topEntries = await readdir(TRANSCRIPTS_DIR, { - withFileTypes: true, - }).catch(() => [] as Dirent[]); - for (const e of topEntries) { - if (e.isDirectory()) { - if (!channelConfigs.has(e.name)) { - await rm(path.join(TRANSCRIPTS_DIR, e.name), { - recursive: true, - force: true, - }); - } - } else if (e.isFile()) { - await rm(path.join(TRANSCRIPTS_DIR, e.name), { force: true }); - } - } - - console.log( - `Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`, - ); - - // Stream LMDB (reverse composite-key order = newest-date first) to emit - // paginated summaries. - const pageSize = SUMMARIES_PAGE_SIZE; - const channelCounts = new Map<string, number>(); - // Seed channel list from config so empty channels still appear in the UI. - for (const cfg of channelConfigs.values()) { - if (cfg.name) channelCounts.set(cfg.name, 0); - } - let pageIndex = 0; - let buffer: DisplaySummary[] = []; - let total = 0; - - const flushPage = async () => { - if (buffer.length === 0) return; - const p = path.join(SUMMARIES_DIR, pageFileName(pageIndex)); - await writeJsonAtomic(p, buffer); - buffer = []; - pageIndex++; - }; - - for (const { value } of sums.getRange({ reverse: true })) { - const s = value as TranscriptSummary; - if (s.channel) { - channelCounts.set(s.channel, (channelCounts.get(s.channel) ?? 0) + 1); - } - buffer.push(toDisplaySummary(s)); - total++; - if (buffer.length >= pageSize) await flushPage(); - } - await flushPage(); - - // Prune stale page files. - const expectedPages = new Set<string>(); - for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i)); - const existing = await readdir(SUMMARIES_DIR).catch(() => [] as string[]); - for (const name of existing) { - if (!name.startsWith("page-")) continue; - if (!expectedPages.has(name)) { - await rm(path.join(SUMMARIES_DIR, name), { force: true }); - } - } - - const channelList: ChannelEntry[] = Array.from(channelCounts.entries()) - .map(([name, count]) => ({ name, count })) - .sort((a, b) => a.name.localeCompare(b.name)); - - const manifest: Manifest = { - version: MANIFEST_VERSION, - totalCount: total, - pageSize, - pageCount: pageIndex, - generatedAt: new Date().toISOString(), - channels: channelList, - }; - await writeJsonAtomic(MANIFEST_PATH, manifest); - - await root.close(); - - const secs = ((Date.now() - t0) / 1000).toFixed(2); - console.log( - `Done in ${secs}s. ${total} transcripts across ${pageIndex} pages; ${channelList.length} channels.`, - ); -} - -main().catch((err) => { - console.error(err); - process.exit(1); -}); diff --git a/package.json b/package.json @@ -4,7 +4,7 @@ "private": true, "type": "module", "scripts": { - "build:index": "pnpm --filter export run build:index", + "build:index": "pnpm --filter yt-dlp-transcript-common exec tsx bin/build-index.ts", "build:export": "pnpm --filter export run build", "build": "pnpm --filter export run build", "start:export": "pnpm --filter export run start", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -14,15 +14,27 @@ importers: common: dependencies: + '@sindresorhus/slugify': + specifier: ^3.0.0 + version: 3.0.0 '@tanstack/react-query': specifier: ^5.99.1 version: 5.100.5(react@19.2.4) + execa: + specifier: ^9.6.1 + version: 9.6.1 fs-extra: specifier: ^11.3.4 version: 11.3.4 + lmdb: + specifier: ^3.5.4 + version: 3.5.4 next: specifier: 16.2.3 version: 16.2.3(@babel/core@7.29.0)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + p-limit: + specifier: ^7.3.0 + version: 7.3.0 react: specifier: 19.2.4 version: 19.2.4 @@ -45,30 +57,21 @@ importers: '@types/react-dom': specifier: ^19.2.3 version: 19.2.3(@types/react@19.2.14) + tsx: + specifier: ^4.21.0 + version: 4.21.0 typescript: specifier: ^5.9.3 version: 5.9.3 export: dependencies: - '@sindresorhus/slugify': - specifier: ^3.0.0 - version: 3.0.0 '@tanstack/react-query': specifier: ^5.99.1 version: 5.100.5(react@19.2.4) - execa: - specifier: ^9.6.1 - version: 9.6.1 - lmdb: - specifier: ^3.5.4 - version: 3.5.4 next: specifier: 16.2.3 version: 16.2.3(@babel/core@7.29.0)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) - p-limit: - specifier: ^7.3.0 - version: 7.3.0 react: specifier: 19.2.4 version: 19.2.4 @@ -88,9 +91,6 @@ importers: '@tailwindcss/postcss': specifier: ^4.2.2 version: 4.2.4 - '@types/fs-extra': - specifier: ^11.0.4 - version: 11.0.4 '@types/node': specifier: ^20.19.39 version: 20.19.39 @@ -106,9 +106,6 @@ importers: eslint-config-next: specifier: 16.2.3 version: 16.2.3(@typescript-eslint/parser@8.59.0(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3))(eslint@9.39.4(jiti@2.6.1))(typescript@5.9.3) - fs-extra: - specifier: ^11.3.4 - version: 11.3.4 tailwindcss: specifier: ^4.2.2 version: 4.2.4