Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 51a1b8f02c7dd7c5cea46020e8ef10164987d5b7
parent 71704bbfef276badb6308a505ddfeb925c775fe2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri,  3 Jul 2026 04:01:13 -0400

Add downloadable transcript & live-chat archive zips

Generate each site's transcript and live-chat archives during its build
into the served public/archives dir, expose them on a new Downloads page
(+ header/footer links), and add a per-video single-file download.

- zip is now the default archive format (was tar.gz); Build-page help copy
  tracks the selected format.
- archive controllers take an optional outDir; compose-site builds per-channel
  and combined zips scoped to the site's members, writes a size-aware
  manifest.json, and drops any file over the cap (default 25 MB, configurable;
  0 = no cap) so a Cloudflare Pages deploy isn't rejected.
- generation is on by default with three opt-out levels: global setting,
  per-site site.json flag (+ archiveMaxBytes), and a per-build BUILD_ARCHIVES=0.
- Downloads page reads the manifest, listing whole-site and per-channel bundles
  with sizes/counts; oversize entries render as unavailable.
- player toolbar gains a per-video download (.txt/.srt/.json) built client-side
  from loaded cues, for whichever of transcript/live-chat is shown.

Verified: real compose-site run generates zips + manifest with correct
oversize handling; outDir produces a valid zip; cue-serializer unit tests pass;
common/editor/export all typecheck clean.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
M.gitignore | 2++
Mcommon/bin/compose-site.ts | 214++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/components/PlayerProvider.tsx | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/components/TranscriptModal.tsx | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/archiveLiveChat.ts | 16+++++++++++-----
Mcommon/controller/archiveTranscripts.ts | 16+++++++++++-----
Mcommon/lib/archiveOptions.ts | 39+++++++++++++++++++++++++++++++++++++--
Mcommon/lib/settings.ts | 10++++++++++
Mcommon/lib/site.ts | 26++++++++++++++++++++++++++
Acommon/lib/vtt.test.ts | 39+++++++++++++++++++++++++++++++++++++++
Mcommon/lib/vtt.ts | 28++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/build/buildAction.ts | 11+++++++++--
Meditor/app/build/components/BuildButtons.tsx | 8++++----
Meditor/app/deploy/buildDeployCore.ts | 9++++++++-
Meditor/app/deploy/components/BuildDeployButton.tsx | 20++++++++++++++++++--
Meditor/app/deploy/components/BuildExportButton.tsx | 17++++++++++++++++-
Meditor/app/settings/actions.ts | 2++
Meditor/app/settings/components/SettingsForm.tsx | 19+++++++++++++++++++
Meditor/app/sites/actions.ts | 17+++++++++++++++++
Meditor/app/sites/components/SiteForm.tsx | 31++++++++++++++++++++++++++++++-
Mexport/CHANGELOG.md | 4++++
Mexport/app/components/Footer.tsx | 11+++++++++++
Mexport/app/components/Header.tsx | 10++++++++++
Aexport/app/downloads/page.tsx | 154+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aexport/app/lib/archives.ts | 30++++++++++++++++++++++++++++++
26 files changed, 820 insertions(+), 25 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -60,6 +60,8 @@ yarn-error.log* /export/public/_headers /export/public/sw.js /export/public/hub-sites.json +# per-site bulk-download archive zips + manifest (regenerate with `pnpm build`) +/export/public/archives/ /export/.export-index/ /transcripts/index.mdb/ diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts @@ -14,16 +14,31 @@ // only that site's data. Checked-in static assets in public/ are left intact. import path from "node:path"; -import { cp, mkdir, rm, readdir, access, readFile, writeFile } from "node:fs/promises"; +import { cp, mkdir, rm, readdir, access, readFile, writeFile, stat } from "node:fs/promises"; import { getPaths } from "../lib/paths"; import { getSite, resolveSocialLinks, resolveHubUrl, type Site } from "../lib/site"; +import { getSettings } from "../lib/settings"; import { DUPLICATES_FILENAME, filterClusterToChannels, type DuplicateReport, } from "../lib/duplicates"; -import type { Manifest } from "../lib/manifest"; +import type { Manifest, SubsManifest } from "../lib/manifest"; import { buildSiteDescriptor } from "../lib/siteDescriptor"; +import { + archiveTranscripts, + archiveCombinedTranscripts, +} from "../controller/archiveTranscripts"; +import { + archiveLiveChat, + archiveCombinedLiveChat, +} from "../controller/archiveLiveChat"; +import { + ARCHIVE_MANIFEST_FILENAME, + DEFAULT_ARCHIVE_MAX_BYTES, + type ArchiveManifest, + type ArchiveManifestEntry, +} from "../lib/archiveOptions"; // CORS + cache headers for Cloudflare Pages (a static `_headers` file at the // deploy root). All served data is public static JSON with no credentials, so @@ -42,6 +57,8 @@ const CORS_HEADERS = `# Generated by compose-site.ts — do not edit by hand. Access-Control-Allow-Origin: * /stats/* Access-Control-Allow-Origin: * +/archives/* + Access-Control-Allow-Origin: * `; // Emit the public federation contract: /site.json (branding + channels + @@ -88,6 +105,196 @@ async function composeServiceWorker( if (await exists(src)) await cp(src, dest); } +// Whether this build generates the downloadable archive zips. Opt-OUT at three +// levels, all defaulting on: the global setting, the per-site `archives` flag, +// and a per-build BUILD_ARCHIVES=0 env (set by the editor's "Skip archive zips" +// control). Effective = AND of all three. +function archivesEnabled(site: Site): boolean { + const globalOn = getSettings().buildArchives !== false; + const siteOn = site.archives !== false; + const buildOn = process.env.BUILD_ARCHIVES !== "0"; + return globalOn && siteOn && buildOn; +} + +// The served-file size cap for this site: per-site override, else MAX_ARCHIVE_BYTES +// env, else the Cloudflare-safe default. 0 = no cap. +function archiveMaxBytes(site: Site): number { + if (typeof site.archiveMaxBytes === "number" && site.archiveMaxBytes >= 0) { + return site.archiveMaxBytes; + } + const env = Number(process.env.MAX_ARCHIVE_BYTES); + if (Number.isFinite(env) && env >= 0) return env; + return DEFAULT_ARCHIVE_MAX_BYTES; +} + +function humanBytes(n: number): string { + if (n < 1024) return `${n} B`; + const units = ["KB", "MB", "GB"]; + let v = n / 1024; + let i = 0; + while (v >= 1024 && i < units.length - 1) { + v /= 1024; + i++; + } + return `${v.toFixed(v >= 10 || i === 0 ? 0 : 1)} ${units[i]}`; +} + +// Channel display names keyed by slug, read from the already-composed per-site +// subs manifest. Slugs are stable; display names are the human-facing titles. +async function channelTitles( + paths: ReturnType<typeof getPaths>, +): Promise<Map<string, string>> { + const titles = new Map<string, string>(); + const manifestPath = path.join(paths.exportSubsDir, "manifest.json"); + if (!(await exists(manifestPath))) return titles; + try { + const sm = JSON.parse(await readFile(manifestPath, "utf8")) as SubsManifest; + for (const c of sm.channels ?? []) { + if (c.slug && c.name) titles.set(c.slug, c.name); + } + } catch { + // A malformed subs manifest just means we fall back to slugs. + } + return titles; +} + +// Stat a produced archive, and enforce the size cap: an oversize file is removed +// from the served dir (so a capped host won't reject the whole deploy) and marked +// oversize in the manifest, which the Downloads page renders as unavailable. +async function describeArchive( + kind: ArchiveManifestEntry["kind"], + scope: string, + outDir: string, + filename: string, + videoCount: number, + maxBytes: number, + channelTitle?: string, +): Promise<ArchiveManifestEntry> { + const filePath = path.join(outDir, filename); + let bytes = 0; + try { + bytes = (await stat(filePath)).size; + } catch { + bytes = 0; + } + const oversize = maxBytes > 0 && bytes > maxBytes; + if (oversize) { + await rm(filePath, { force: true }); + console.warn( + `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — not served`, + ); + } + const entry: ArchiveManifestEntry = { kind, scope, filename, bytes, videoCount }; + if (channelTitle) entry.channelTitle = channelTitle; + if (oversize) entry.oversize = true; + return entry; +} + +// Generate this site's bulk-download archive zips into public/archives and write +// a manifest.json the Downloads page reads. Scoped to the site's member channels, +// mirroring how every other per-site asset is composed. Runs on every build so +// the zips always match the current corpus. +async function composeArchives( + site: Site, + memberSlugs: string[], + paths: ReturnType<typeof getPaths>, +): Promise<void> { + const outDir = path.join(paths.exportPublicDir, "archives"); + // Always clear first so a disabled site (or one that lost the feature) never + // serves a stale bundle from a previous build. + await rm(outDir, { recursive: true, force: true }); + if (!archivesEnabled(site)) { + console.log("[archives] disabled for this build — skipping."); + return; + } + await mkdir(outDir, { recursive: true }); + + const maxBytes = archiveMaxBytes(site); + const log = (m: string) => { + if (m) console.log(`[archives] ${m}`); + }; + const build = { format: "zip" as const }; + const common = { paths, channelSlugs: memberSlugs, outDir, build, onLog: log }; + + const transcripts = await archiveTranscripts(common); + const transcriptsAll = await archiveCombinedTranscripts(common); + const liveChat = await archiveLiveChat(common); + const liveChatAll = await archiveCombinedLiveChat(common); + + const titles = await channelTitles(paths); + const bySlug = (a: { slug: string }, b: { slug: string }) => + a.slug.localeCompare(b.slug); + const entries: ArchiveManifestEntry[] = []; + + // Transcripts: combined ("all") first, then per-channel by slug. + if (transcriptsAll.archivePath) { + entries.push( + await describeArchive( + "transcripts", + "all", + outDir, + path.basename(transcriptsAll.archivePath), + transcriptsAll.totalTranscripts, + maxBytes, + ), + ); + } + for (const a of [...transcripts.archives].sort(bySlug)) { + entries.push( + await describeArchive( + "transcripts", + a.slug, + outDir, + path.basename(a.archivePath), + a.transcriptCount, + maxBytes, + titles.get(a.slug), + ), + ); + } + + // Live chat: same ordering. Channels with no chat produce nothing to list. + if (liveChatAll.archivePath) { + entries.push( + await describeArchive( + "live-chat", + "all", + outDir, + path.basename(liveChatAll.archivePath), + liveChatAll.totalLiveChats, + maxBytes, + ), + ); + } + for (const a of [...liveChat.archives].sort(bySlug)) { + entries.push( + await describeArchive( + "live-chat", + a.slug, + outDir, + path.basename(a.archivePath), + a.liveChatCount, + maxBytes, + titles.get(a.slug), + ), + ); + } + + const manifest: ArchiveManifest = { + version: 1, + thresholdBytes: maxBytes, + entries, + }; + await writeFile( + path.join(outDir, ARCHIVE_MANIFEST_FILENAME), + JSON.stringify(manifest, null, 2) + "\n", + ); + const served = entries.filter((e) => !e.oversize).length; + console.log( + `[archives] wrote ${served}/${entries.length} archive(s) to public/archives.`, + ); +} + async function exists(p: string): Promise<boolean> { try { await access(p); @@ -201,6 +408,9 @@ async function main(): Promise<void> { // --- service worker (only when this instance ships a PWA) --- await composeServiceWorker(site, paths); + // --- bulk-download archive zips (on by default; see archivesEnabled) --- + await composeArchives(site, memberSlugs, paths); + const channelDirs = (await readdir(paths.exportTranscriptsDir).catch( () => [] as string[], )).length; diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx @@ -17,6 +17,7 @@ import { fetchSubs } from "./subsCache"; import { useUrlParams, writeUrlParams } from "./urlState"; import type { ModalMode } from "./urlState"; import { formatDate, formatDuration } from "../lib/format"; +import { cuesToSrt, cuesToText } from "../lib/vtt"; import type { Platform } from "../lib/transcripts"; import { vodExpiry } from "../lib/vodExpiry"; import { VodExpiredBadge } from "./badges"; @@ -125,6 +126,9 @@ type PlayerState = { clearClip: () => void; copyDownloadCommand: () => Promise<boolean>; copyShareUrl: () => Promise<boolean>; + // Download the currently-shown transcript or live-chat log (per modalMode) as + // a single file, generated client-side from the loaded cues. + downloadTranscriptFile: (fmt: "json" | "txt" | "srt") => void; openInPreservetube: () => void; seekTo: (seconds: number) => void; }; @@ -418,6 +422,53 @@ export function PlayerProvider({ ); }, [data]); + // Save the currently-shown cues (transcript or live chat, per modalMode) as a + // single file. Built entirely from cues already loaded in the browser — no + // server round-trip and no dependency on the bulk archive zips. + const downloadTranscriptFile = useCallback( + (fmt: "json" | "txt" | "srt") => { + const isChat = modalMode === "chat"; + const cues = isChat + ? chat.slug === activeSlug + ? chat.cues + : null + : (data?.cues ?? null); + if (!data || !cues || cues.length === 0) return; + + let body: string; + let mime: string; + if (fmt === "json") { + body = JSON.stringify(cues, null, 2); + mime = "application/json"; + } else if (fmt === "srt") { + body = cuesToSrt(cues); + mime = "application/x-subrip"; + } else { + body = cuesToText(cues); + mime = "text/plain"; + } + + const kind = isChat ? "live-chat" : "transcript"; + const base = + (data.slug || data.title || kind) + .replace(/[^a-z0-9]+/gi, "-") + .replace(/^-+|-+$/g, "") + .toLowerCase() || kind; + const filename = `${base}.${kind}.${fmt}`; + + const blob = new Blob([body], { type: `${mime};charset=utf-8` }); + const url = URL.createObjectURL(blob); + const a = document.createElement("a"); + a.href = url; + a.download = filename; + document.body.appendChild(a); + a.click(); + a.remove(); + URL.revokeObjectURL(url); + }, + [data, chat, activeSlug, modalMode], + ); + const seekTo = useCallback((seconds: number) => { if (playerRef.current) { playerRef.current.seekTo(seconds, "seconds"); @@ -614,6 +665,7 @@ export function PlayerProvider({ clearClip, copyDownloadCommand, copyShareUrl, + downloadTranscriptFile, openInPreservetube, seekTo, }), @@ -636,6 +688,7 @@ export function PlayerProvider({ clearClip, copyDownloadCommand, copyShareUrl, + downloadTranscriptFile, openInPreservetube, seekTo, ], diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx @@ -41,6 +41,7 @@ export default function TranscriptModal() { clearClip, copyDownloadCommand, copyShareUrl, + downloadTranscriptFile, openInPreservetube, seekTo, } = usePlayer(); @@ -50,6 +51,8 @@ export default function TranscriptModal() { const copyResetRef = useRef<number | null>(null); const [shareCopied, setShareCopied] = useState(false); const shareResetRef = useRef<number | null>(null); + const [downloadOpen, setDownloadOpen] = useState(false); + const downloadRef = useRef<HTMLDivElement | null>(null); // Tracks whether the next scrollToIndex should animate. Smooth on // user-initiated changes (cue click, mode toggle, initial open); instant // on natural playhead drift so 4 Hz progress ticks don't keep retriggering @@ -57,6 +60,31 @@ export default function TranscriptModal() { const scrollKindRef = useRef<"smooth" | "auto">("smooth"); const isChat = modalMode === "chat"; + const canDownloadFile = isChat + ? (chatCues?.length ?? 0) > 0 + : (data?.cues?.length ?? 0) > 0; + + // Close the download format menu on an outside click or Escape. + useEffect(() => { + if (!downloadOpen) return; + const onDown = (e: MouseEvent) => { + if ( + downloadRef.current && + !downloadRef.current.contains(e.target as Node) + ) { + setDownloadOpen(false); + } + }; + const onKey = (e: KeyboardEvent) => { + if (e.key === "Escape") setDownloadOpen(false); + }; + document.addEventListener("mousedown", onDown); + document.addEventListener("keydown", onKey); + return () => { + document.removeEventListener("mousedown", onDown); + document.removeEventListener("keydown", onKey); + }; + }, [downloadOpen]); // Pre-compute author/body split once per cue array so the render hot path // doesn't redo `indexOf`/`slice` on every progress tick. Memo key is the @@ -225,6 +253,36 @@ export default function TranscriptModal() { char={shareCopied ? "✓" : "⤴"} disabled={!data} /> + <div className="relative" ref={downloadRef}> + <ControlButton + title={ + canDownloadFile + ? `Download this ${isChat ? "live chat" : "transcript"} as a file` + : "Nothing to download yet" + } + onClick={() => setDownloadOpen((v) => !v)} + char="⤓" + disabled={!canDownloadFile} + highlight={downloadOpen} + /> + {downloadOpen && ( + <div className="absolute right-0 top-9 z-10 flex flex-col overflow-hidden rounded-md bg-zinc-900 shadow-xl ring-1 ring-white/15"> + {(["txt", "srt", "json"] as const).map((fmt) => ( + <button + key={fmt} + type="button" + onClick={() => { + downloadTranscriptFile(fmt); + setDownloadOpen(false); + }} + className="px-3 py-1.5 text-left font-mono text-xs uppercase tracking-wide text-zinc-200 hover:bg-white/10" + > + .{fmt} + </button> + ))} + </div> + )} + </div> {data?.platform === "youtube" && ( <ControlButton title="Open in Preservetube" diff --git a/common/controller/archiveLiveChat.ts b/common/controller/archiveLiveChat.ts @@ -46,6 +46,10 @@ export type ArchiveLiveChatOptions = { concurrency?: number; channelSlugs?: string[]; build?: Partial<ArchiveBuildOptions>; + // Destination for the finished archives. Defaults to transcripts/export/ + // archives; compose-site passes the served public/archives dir. Mirrors + // ArchiveTranscriptsOptions.outDir. + outDir?: string; }; export type ArchiveLiveChatResult = { @@ -185,7 +189,8 @@ export async function archiveLiveChat( `Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`, ); - await mkdir(archivesDir(opts.paths), { recursive: true }); + const outDir = opts.outDir ?? archivesDir(opts.paths); + await mkdir(outDir, { recursive: true }); await mkdir(stagingDir(opts.paths), { recursive: true }); const allChannels = await listChannels(opts.paths); @@ -200,7 +205,7 @@ export async function archiveLiveChat( } const staging = path.join(stagingDir(opts.paths), ch.slug); const archivePath = path.join( - archivesDir(opts.paths), + outDir, `${ch.slug}.${ARCHIVE_INFIX}.${archiveExtension(build.format)}`, ); try { @@ -244,7 +249,7 @@ export async function archiveLiveChat( log( `Wrote ${archives.length} archive${archives.length === 1 ? "" : "s"} containing ${totalLiveChats} live chats to:`, ); - log(` ${archivesDir(opts.paths)}`); + log(` ${outDir}`); for (const a of archives) { log(` - ${path.basename(a.archivePath)} (${a.liveChatCount} live chats)`); } @@ -271,7 +276,8 @@ export async function archiveCombinedLiveChat( `Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`, ); - await mkdir(archivesDir(opts.paths), { recursive: true }); + const outDir = opts.outDir ?? archivesDir(opts.paths); + await mkdir(outDir, { recursive: true }); await mkdir(stagingDir(opts.paths), { recursive: true }); const combinedStaging = path.join( @@ -279,7 +285,7 @@ export async function archiveCombinedLiveChat( COMBINED_STAGING_DIR, ); const archivePath = path.join( - archivesDir(opts.paths), + outDir, `${COMBINED_BASENAME}.${archiveExtension(build.format)}`, ); diff --git a/common/controller/archiveTranscripts.ts b/common/controller/archiveTranscripts.ts @@ -60,6 +60,10 @@ export type ArchiveTranscriptsOptions = { concurrency?: number; channelSlugs?: string[]; build?: Partial<ArchiveBuildOptions>; + // Destination for the finished archives. Defaults to transcripts/export/ + // archives (the standalone editor actions' location); compose-site passes the + // served public/archives dir so the zips ship with the site build. + outDir?: string; }; export type ArchiveTranscriptsResult = { @@ -281,7 +285,8 @@ export async function archiveTranscripts( `Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`, ); - await mkdir(archivesDir(opts.paths), { recursive: true }); + const outDir = opts.outDir ?? archivesDir(opts.paths); + await mkdir(outDir, { recursive: true }); await mkdir(stagingDir(opts.paths), { recursive: true }); const allChannels = await listChannels(opts.paths); @@ -296,7 +301,7 @@ export async function archiveTranscripts( } const staging = path.join(stagingDir(opts.paths), ch.slug); const archivePath = path.join( - archivesDir(opts.paths), + outDir, `${ch.slug}.${archiveExtension(build.format)}`, ); try { @@ -340,7 +345,7 @@ export async function archiveTranscripts( log( `Wrote ${archives.length} archive${archives.length === 1 ? "" : "s"} containing ${totalTranscripts} transcripts to:`, ); - log(` ${archivesDir(opts.paths)}`); + log(` ${outDir}`); for (const a of archives) { log(` - ${path.basename(a.archivePath)} (${a.transcriptCount} transcripts)`); } @@ -367,12 +372,13 @@ export async function archiveCombinedTranscripts( `Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`, ); - await mkdir(archivesDir(opts.paths), { recursive: true }); + const outDir = opts.outDir ?? archivesDir(opts.paths); + await mkdir(outDir, { recursive: true }); await mkdir(stagingDir(opts.paths), { recursive: true }); const combinedStaging = path.join(stagingDir(opts.paths), COMBINED_STAGING_DIR); const archivePath = path.join( - archivesDir(opts.paths), + outDir, `${COMBINED_BASENAME}.${archiveExtension(build.format)}`, ); diff --git a/common/lib/archiveOptions.ts b/common/lib/archiveOptions.ts @@ -5,10 +5,11 @@ export type ArchiveFormat = "tar.gz" | "tar.xz" | "zip"; +// `zip` first so it leads the editor's format dropdown, matching the default. export const ARCHIVE_FORMATS: ReadonlyArray<ArchiveFormat> = [ + "zip", "tar.gz", "tar.xz", - "zip", ]; export type ArchiveBuildOptions = { @@ -24,8 +25,42 @@ export type ArchiveBuildOptions = { }; export const DEFAULT_ARCHIVE_OPTIONS: ArchiveBuildOptions = { - format: "tar.gz", + format: "zip", compressionLevel: 6, includeMetadata: true, prettyPrint: false, }; + +// --- Served archive manifest ------------------------------------------------- +// compose-site.ts writes public/archives/manifest.json describing the bulk +// download zips it produced for a site. The Downloads page reads it to list the +// archives with sizes/counts without hardcoding filenames. Client-safe (no fs). + +export type ArchiveManifestEntry = { + kind: "transcripts" | "live-chat"; + // "all" for the combined archive, otherwise a channel slug. + scope: "all" | string; + // Present when scope is a channel slug. + channelTitle?: string; + // Filename relative to /archives/ (e.g. "all-transcripts.zip", "grumps.zip"). + filename: string; + bytes: number; + videoCount: number; + // True when the file exceeded thresholdBytes: it is NOT served (removed so a + // size-capped host like Cloudflare Pages won't reject the whole deploy) and + // the UI renders it as unavailable rather than a broken link. + oversize?: boolean; +}; + +export type ArchiveManifest = { + version: 1; + // The size cap applied this build, so the UI can explain omissions. 0 = no cap. + thresholdBytes: number; + entries: ArchiveManifestEntry[]; +}; + +export const ARCHIVE_MANIFEST_FILENAME = "manifest.json"; + +// Default per-file size cap. Cloudflare Pages rejects any single asset > 25 MB, +// which is why oversize combined archives must be dropped from what's served. +export const DEFAULT_ARCHIVE_MAX_BYTES = 25 * 1024 * 1024; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -97,6 +97,11 @@ export type SiteSettings = { // video once the stream ends. Per-channel override available // (ChannelConfig.skipLiveDownloads). skipLiveDownloads: boolean; + // Whether site builds generate downloadable transcript/live-chat archive zips + // (into public/archives, linked on the Downloads page). Global default; a site + // can opt out via site.json `archives: false`, and a single build can skip via + // the "Skip archive zips" build control. Opt-out: default true. + buildArchives: boolean; // Debounce preset for the global snapshot scheduler: how long it waits after // the last report-changing action before regenerating affected channel // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap). @@ -449,6 +454,7 @@ function defaults(): SiteSettings { parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback: false, skipLiveDownloads: true, + buildArchives: true, reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET, autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS, syncScheduler: defaultSyncScheduler(), @@ -637,6 +643,9 @@ export function getSettings(): SiteSettings { if (typeof merged.skipLiveDownloads !== "boolean") { merged.skipLiveDownloads = true; } + if (typeof merged.buildArchives !== "boolean") { + merged.buildArchives = true; + } if (!isReportDebouncePreset(merged.reportDebouncePreset)) { merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET; } @@ -813,6 +822,7 @@ export async function writeSettings(next: SiteSettings): Promise<void> { ), inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true, skipLiveDownloads: next.skipLiveDownloads !== false, + buildArchives: next.buildArchives !== false, reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset) ? next.reportDebouncePreset : DEFAULT_REPORT_DEBOUNCE_PRESET, diff --git a/common/lib/site.ts b/common/lib/site.ts @@ -76,6 +76,17 @@ export type Site = { // trusts only the hub PWA. Set true to make this site its own installable PWA. // See export/app/lib/mode.ts (shipsPwa) and common/lib/siteDescriptor.ts. pwa?: boolean; + // Whether the site build generates downloadable transcript/live-chat archive + // zips into public/archives (and links them on the Downloads page). This is + // an opt-OUT: undefined/true = on, only explicit `false` disables. Also gated + // by the global setting and a per-build flag (see compose-site.ts). Default-on + // because bulk download is the point of publishing a corpus. + archives?: boolean; + // Per-site override for the served-file size cap (bytes). Any archive larger + // than this is dropped from what's served and flagged in the manifest so a + // capped host (Cloudflare Pages: 25 MB) won't reject the deploy. 0 = no cap. + // Absent = the global DEFAULT_ARCHIVE_MAX_BYTES / MAX_ARCHIVE_BYTES env. + archiveMaxBytes?: number; // Per-site override for the hub this site belongs under (the PWA it points // visitors toward). Absent = inherit the family default SiteSettings.homepageUrl. // Resolve with resolveHubUrl(). Surfaced on /site.json so a hub can tell member @@ -246,6 +257,14 @@ export function parseSite(siteId: string, raw: unknown): Site { siteUrl: parseSiteUrl(r.siteUrl), relatedSites: parseRelatedSites(r.relatedSites), pwa: r.pwa === true, + // Opt-out: only an explicit false disables. Absent/true stays on. + archives: r.archives !== false, + archiveMaxBytes: + typeof r.archiveMaxBytes === "number" && + Number.isFinite(r.archiveMaxBytes) && + r.archiveMaxBytes >= 0 + ? Math.floor(r.archiveMaxBytes) + : undefined, hubUrl: parseSiteUrl(r.hubUrl), }; } @@ -408,6 +427,13 @@ export async function writeSite( ? { relatedSites: parseRelatedSites(site.relatedSites) } : {}), ...(site.pwa ? { pwa: true } : {}), + // Persist only the non-default: archives is on unless explicitly disabled. + ...(site.archives === false ? { archives: false } : {}), + ...(typeof site.archiveMaxBytes === "number" && + Number.isFinite(site.archiveMaxBytes) && + site.archiveMaxBytes >= 0 + ? { archiveMaxBytes: Math.floor(site.archiveMaxBytes) } + : {}), ...(parseSiteUrl(site.hubUrl) ? { hubUrl: parseSiteUrl(site.hubUrl) } : {}), }; const dir = siteDir(paths, site.siteId); diff --git a/common/lib/vtt.test.ts b/common/lib/vtt.test.ts @@ -0,0 +1,39 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { cuesToText, cuesToSrt, type Cue } from "./vtt"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/vtt.test.ts + +const cues: Cue[] = [ + { start: 0, end: 2.5, text: "Hello there" }, + { start: 3661.25, end: 3663, text: "General Kenobi" }, +]; + +test("cuesToText writes one line per cue, trailing newline", () => { + assert.equal(cuesToText(cues), "Hello there\nGeneral Kenobi\n"); +}); + +test("cuesToText on empty cues is a lone newline", () => { + assert.equal(cuesToText([]), "\n"); +}); + +test("cuesToSrt numbers blocks and formats HH:MM:SS,mmm timing", () => { + const srt = cuesToSrt(cues); + assert.equal( + srt, + "1\n00:00:00,000 --> 00:00:02,500\nHello there\n" + + "\n" + + "2\n01:01:01,250 --> 01:01:03,000\nGeneral Kenobi\n", + ); +}); + +test("cuesToSrt falls back to start when end is missing", () => { + const noEnd = [{ start: 5, end: undefined as unknown as number, text: "x" }]; + const srt = cuesToSrt(noEnd); + assert.equal(srt, "1\n00:00:05,000 --> 00:00:05,000\nx\n"); +}); + +test("cuesToSrt clamps negative times to zero", () => { + const neg = [{ start: -3, end: -1, text: "neg" }]; + assert.equal(cuesToSrt(neg), "1\n00:00:00,000 --> 00:00:00,000\nneg\n"); +}); diff --git a/common/lib/vtt.ts b/common/lib/vtt.ts @@ -74,6 +74,34 @@ export function parseVtt(src: string): Cue[] { return deduped; } +// One text line per cue. For live chat each cue's text is already "author: +// message", so this doubles as a readable chat log. +export function cuesToText(cues: Cue[]): string { + return cues.map((c) => c.text).join("\n") + "\n"; +} + +function srtTime(seconds: number): string { + const s = Math.max(0, seconds); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + const sec = Math.floor(s % 60); + const ms = Math.round((s - Math.floor(s)) * 1000); + const pad = (n: number, w = 2) => String(n).padStart(w, "0"); + return `${pad(h)}:${pad(m)}:${pad(sec)},${pad(ms, 3)}`; +} + +// SubRip (.srt): numbered, timestamped cue blocks. `end` falls back to `start` +// so zero-length cues (some chat lines) still produce valid timing. +export function cuesToSrt(cues: Cue[]): string { + return cues + .map((c, i) => { + const start = srtTime(c.start); + const end = srtTime(c.end ?? c.start); + return `${i + 1}\n${start} --> ${end}\n${c.text}\n`; + }) + .join("\n"); +} + export function formatTimestamp(seconds: number): string { const s = Math.floor(seconds); const h = Math.floor(s / 3600); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Every site build now bundles downloadable transcript & live-chat archive zips.** The archive builders (per-channel `<slug>.zip`, combined `all-transcripts.zip` / `all-live-chat.zip`) previously only ran as standalone actions that wrote to a non-served directory; now `compose-site` generates them for the site's own channels straight into the served `public/archives/` and writes a `manifest.json` (sizes + counts) that the site's new **Downloads** page reads. **`zip` is now the default archive format** everywhere (was `tar.gz`), and the Build page's format help text tracks the selected format. Generation is **on by default with three opt-out levels**: a global **Generate archive zips on build** toggle in Settings, a per-site **Generate archive zips** toggle (plus an optional **Archive size cap (MB)**) on the site's page, and a per-build **Skip archive zips** checkbox on the Build and Build & Deploy controls (`BUILD_ARCHIVES=0`). Because a single file over ~25 MB breaks a Cloudflare Pages deploy, any archive over the cap (default 25 MB; `0` = no cap) is dropped from what's served and flagged `oversize` in the manifest so the deploy still succeeds and the Downloads page shows it as unavailable rather than a dead link. See `common/bin/compose-site.ts` (`composeArchives`), `common/controller/archive{Transcripts,LiveChat}.ts` (new `outDir` option), `common/lib/archiveOptions.ts` (default + manifest types), `common/lib/{site,settings}.ts` (opt-out flags), and `editor/app/{deploy/buildDeployCore.ts,build/buildAction.ts,deploy/components/Build{Export,Deploy}Button.tsx,sites/components/SiteForm.tsx,settings/components/SettingsForm.tsx}`. - **You can now change a channel's slug (its id) — deliberately, from the Danger zone.** A channel's slug *is* its on-disk directory name (`transcripts/channels/<slug>/`), so it used to be fixed at creation ("Slug is fixed once a channel is created"). A new **Rename** form in the channel's Danger zone lifts that: enter a new slug and **type the current slug to confirm** (same friction as delete), and the rename is blocked while the channel has running/queued jobs (the in-memory registry keys by slug). Because the slug is a directory name, the rename does a **full migration** of every slug-keyed store so nothing silently breaks: it moves the channel dir (config, data, playlist, snapshot, shards, failed lists) **and** the saved-video store dir — rewriting each `saved-video.json` pointer's absolute `dir` so persisted source videos still resolve — then retargets every site.json membership, the sync scheduler's per-channel backoff state, and any job bookmarks. The two filesystem moves run first and roll back on failure; the metadata updates that follow are atomic and best-effort (surfaced as warnings). Renaming **changes the channel's public URL** (the old one 404s), which the form warns about. The slug grammar is also now validated on create. See `common/controller/renameChannel.ts`, `common/controller/channels.ts` (`isValidChannelSlug`), `common/lib/savedVideo-server.ts` (`rewriteSavedVideoDir`), `common/jobs/bookmarks.ts` (`renameChannelInBookmarks`), `editor/app/channels/{actions.ts,components/RenameChannelForm.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-rename.spec.ts`. - **New Queue diagnostics page (`/jobs/queue`): see & force-release stuck jobs.** The job system has two sources of truth that can drift — the registry owns each job's `status`, the scheduler owns the running SLOT per queue. A cancel that never finalizes (a child that ignored SIGTERM, a crashed finalizer) leaves a job "cancelled" in the registry while the scheduler still marks its slot running, silently blocking every job behind it on that queue — and the Active Jobs page hides it (it filters to running/queued). The new **Queue** page reconciles the two: it builds from the **scheduler** as the source of truth for slots, cross-checks each against its registry record, and flags a running head as **stuck** when the record is terminal-but-holding-slot, evicted, or (softer) a live job idle past 10 minutes. It **auto-heals** the hard cases on every view/poll (frees terminal/evicted slots), shows a health strip (active queues, running, queued, **stuck**, workers), per-queue cards with the held-for duration / PID (`kill -9` hint) / last log line, and a **Force-release** button per slot (SIGKILLs the child and frees the slot unconditionally) plus a **Reap all stuck** action. Force-release is also available on any running job in Active Jobs, and Active Jobs links to Queue with a stuck-count badge. See `common/jobs/registry.ts` (`forceRelease`), `editor/app/jobs/queue/*`, `editor/app/jobs/{actions.ts,components/ForceReleaseJobButton.tsx}`, and `editor/e2e/queue.spec.ts`. - **Jobs page: real log retention + pagination (replaces the dead "Clear archived logs" button).** The old button only deleted logs absent from the in-memory registry — which, since the registry keeps the 100 newest finished jobs and sidecars preserve their real status, was almost never anything, so it did nothing. It's replaced by a **Clear logs** dropdown that prunes finished-job logs by age (older than 7 / 30 / 90 days) or all at once; running/queued jobs are never deleted. The `.jobs` directory also **self-trims on job finish** (throttled; keep newest 500, drop >30 days) so it can't grow unbounded. Job ids are now **ULIDs** (lexicographically time-sortable, timestamp decodable from the id), letting the list **paginate** — `listAllJobs` returns one page (default 50, grown by a **Load more** link) and only `stat`s/reads the sidecar for the shown page instead of every file on every load. `jobIdTime()` decodes both ULID and the legacy `<t36>-<rand>` ids, so existing on-disk logs still sort/read correctly. See `common/jobs/{ulid,listJobs,registry,streamCommand}.ts`, `editor/app/jobs/{page.tsx,actions.ts,components/ClearLogsMenu.tsx,[id]/page.tsx}`, and `editor/e2e/jobs.spec.ts`. diff --git a/editor/app/build/buildAction.ts b/editor/app/build/buildAction.ts @@ -64,6 +64,7 @@ export async function buildExportAction( siteId: string, queueKey?: string, skipData?: boolean, + skipArchives?: boolean, ): Promise<StreamActionResult> { const paths = getPaths(); const id = siteId.trim(); @@ -81,7 +82,10 @@ export async function buildExportAction( queueKey: queueKey === undefined ? DEFAULT_BUILD_QUEUE : queueKey.trim(), paths, fn: async (onLog, signal) => { - const code = await runBuildPhase(onLog, signal, id, paths, { skipData }); + const code = await runBuildPhase(onLog, signal, id, paths, { + skipData, + skipArchives, + }); if (signal.aborted) return; if (code !== 0) throw new Error(`Build failed (exit ${code})`); }, @@ -94,6 +98,7 @@ export async function buildExportAction( // deploy queue so it never overlaps a standalone deploy or another build-deploy. export async function buildAndDeployAction( siteId: string, + skipArchives?: boolean, ): Promise<StreamActionResult> { const paths = getPaths(); const id = siteId.trim(); @@ -113,7 +118,9 @@ export async function buildAndDeployAction( paths, fn: async (onLog, signal) => { onLog("=== Build ===\n"); - const buildCode = await runBuildPhase(onLog, signal, id, paths); + const buildCode = await runBuildPhase(onLog, signal, id, paths, { + skipArchives, + }); // A cancel mid-build must NOT proceed to deploy. if (signal.aborted) return; if (buildCode !== 0) { diff --git a/editor/app/build/components/BuildButtons.tsx b/editor/app/build/components/BuildButtons.tsx @@ -140,7 +140,7 @@ export function BuildButtons({ existingQueues }: Props) { <div> <h2 className="text-lg font-semibold">Build transcript archives</h2> <p className="text-sm text-muted-foreground"> - Produces one <code>&lt;channel&gt;.tar.gz</code> per channel + Produces one <code>&lt;channel&gt;.{archiveOptions.format}</code> per channel under <code>transcripts/export/archives/</code>, containing each video&apos;s compact <code>transcript.cues.json</code> plus a{" "} <code>channel.json</code> manifest. Normalizes missing cues on @@ -178,7 +178,7 @@ export function BuildButtons({ existingQueues }: Props) { </h2> <p className="text-sm text-muted-foreground"> Produces a single{" "} - <code>all-transcripts.tar.gz</code> under{" "} + <code>all-transcripts.{combinedOptions.format}</code> under{" "} <code>transcripts/export/archives/</code> containing every channel plus a top-level <code>manifest.json</code> index. Normalizes missing cues on demand. @@ -241,7 +241,7 @@ export function BuildButtons({ existingQueues }: Props) { <div> <h2 className="text-lg font-semibold">Build live chat archives</h2> <p className="text-sm text-muted-foreground"> - Produces one <code>&lt;channel&gt;.live_chat.tar.gz</code> per + Produces one <code>&lt;channel&gt;.live_chat.{liveChatArchiveOptions.format}</code> per channel under <code>transcripts/export/archives/</code>, containing each video&apos;s parsed{" "} <code>live_chat.cues.json</code> plus a{" "} @@ -281,7 +281,7 @@ export function BuildButtons({ existingQueues }: Props) { Build combined live chat archive </h2> <p className="text-sm text-muted-foreground"> - Produces a single <code>all-live-chat.tar.gz</code> under{" "} + Produces a single <code>all-live-chat.{liveChatCombinedOptions.format}</code> under{" "} <code>transcripts/export/archives/</code> containing every channel&apos;s parsed live chats plus a top-level{" "} <code>manifest.json</code> index. Normalizes missing live chats diff --git a/editor/app/deploy/buildDeployCore.ts b/editor/app/deploy/buildDeployCore.ts @@ -30,7 +30,7 @@ export async function runBuildPhase( signal: AbortSignal, siteId: string, paths: Paths, - opts?: { skipData?: boolean }, + opts?: { skipData?: boolean; skipArchives?: boolean }, ): Promise<number> { const { mode } = getSettings().buildPipeline; if (mode === "docker") { @@ -51,6 +51,12 @@ export async function runBuildPhase( "existing .export-index staging.\n", ); } + // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes + // compose-site skip generation this build regardless of the global/site flags. + const skipArchives = opts?.skipArchives === true; + if (skipArchives) { + onLog("[notice] Skipping archive-zip generation for this build.\n"); + } return runChildIntoLog(onLog, signal, { command: "pnpm", args: ["run", skipData ? "build:nodata" : "build"], @@ -61,6 +67,7 @@ export async function runBuildPhase( TRANSCRIPTS_DIR: paths.transcriptsDir, EXPORT_PUBLIC_DIR: paths.exportPublicDir, SITE_ID: siteId, + ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}), }, }); } diff --git a/editor/app/deploy/components/BuildDeployButton.tsx b/editor/app/deploy/components/BuildDeployButton.tsx @@ -1,5 +1,6 @@ "use client"; +import { useState } from "react"; import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; import { buildAndDeployAction } from "../../build/buildAction"; import { cancelJobAction } from "../../jobs/actions"; @@ -20,6 +21,7 @@ export function BuildDeployButton({ cloudflareProject, }: Props) { const ready = Boolean(siteId && cloudflareProject); + const [skipArchives, setSkipArchives] = useState(false); let notice: React.ReactNode; if (!siteId) { @@ -46,13 +48,27 @@ export function BuildDeployButton({ return ( <StreamActionLog - trigger={() => buildAndDeployAction(siteId as string)} + trigger={() => buildAndDeployAction(siteId as string, skipArchives)} cancelAction={cancelJobAction} buttonLabel="Build & deploy" runningLabel="Building & deploying…" label="Build and deploy" disabled={!ready} - extraControls={notice} + extraControls={ + <div className="flex flex-col gap-2"> + {notice} + {ready && ( + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + checked={skipArchives} + onChange={(e) => setSkipArchives(e.target.checked)} + /> + Skip archive zips (transcript &amp; live-chat downloads) + </label> + )} + </div> + } /> ); } diff --git a/editor/app/deploy/components/BuildExportButton.tsx b/editor/app/deploy/components/BuildExportButton.tsx @@ -18,10 +18,13 @@ const DEFAULT_BUILD_QUEUE = "build"; export function BuildExportButton({ existingQueues, siteId, siteTitle }: Props) { const [exportQueue, setExportQueue] = useState(DEFAULT_BUILD_QUEUE); const [skipData, setSkipData] = useState(false); + const [skipArchives, setSkipArchives] = useState(false); return ( <StreamActionLog - trigger={() => buildExportAction(siteId as string, exportQueue, skipData)} + trigger={() => + buildExportAction(siteId as string, exportQueue, skipData, skipArchives) + } cancelAction={cancelJobAction} buttonLabel="Build static export" runningLabel="Building static export…" @@ -50,6 +53,18 @@ export function BuildExportButton({ existingQueues, siteId, siteTitle }: Props) Reuses the existing index/stats — only composes the site and runs next build. Do a full build first if data is stale. </p> + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + checked={skipArchives} + onChange={(e) => setSkipArchives(e.target.checked)} + /> + Skip archive zips (transcript &amp; live-chat downloads) + </label> + <p className="text-xs text-muted-foreground"> + Skips generating the downloadable archive zips for this build only. + Leave off to keep the site&apos;s Downloads page current. + </p> <QueueControl value={exportQueue} onChange={setExportQueue} diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -50,6 +50,7 @@ export async function saveSettingsAction( const inlineTranscribeOnFallback = formData.get("inlineTranscribeOnFallback") === "on"; const skipLiveDownloads = formData.get("skipLiveDownloads") === "on"; + const buildArchives = formData.get("buildArchives") === "on"; const reportDebouncePresetRaw = String( formData.get("reportDebouncePreset") ?? "", ).trim(); @@ -212,6 +213,7 @@ export async function saveSettingsAction( parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback, skipLiveDownloads, + buildArchives, reportDebouncePreset, autoRefreshIntervalSeconds: autoRefreshParsed, syncScheduler, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -163,6 +163,25 @@ export function SettingsForm({ initial, apps }: Props) { </span> </span> </label> + <label className="flex items-start gap-2 text-sm"> + <input + type="checkbox" + name="buildArchives" + defaultChecked={initial.buildArchives} + className="mt-1" + /> + <span className="flex flex-col gap-1"> + <span className="font-medium"> + Generate downloadable archive zips on build + </span> + <span className="text-xs text-muted-foreground"> + Each site build produces transcript &amp; live-chat archive zips (per + channel and combined) and lists them on the site&apos;s Downloads + page. On by default. A site can opt out on its own page, and a single + build can skip them from the Build controls. + </span> + </span> + </label> <label className="flex flex-col gap-1 text-sm"> <span className="font-medium">Report refresh debounce</span> <select diff --git a/editor/app/sites/actions.ts b/editor/app/sites/actions.ts @@ -71,6 +71,21 @@ export async function saveSiteAction( } const hubUrl = parseSiteUrl(hubUrlRaw); const pwa = formData.get("pwa") === "on"; + // Archives default on: an unchecked (default-on) box yields no "archives" key + // → false → persisted as the explicit opt-out. + const archives = formData.get("archives") === "on"; + const archiveMaxMBRaw = String(formData.get("archiveMaxMB") ?? "").trim(); + let archiveMaxBytes: number | undefined; + if (archiveMaxMBRaw) { + const mb = Number(archiveMaxMBRaw); + if (!Number.isFinite(mb) || mb < 0) { + return { + ok: false, + error: "Archive size cap must be a non-negative number of MB, or blank.", + }; + } + archiveMaxBytes = Math.floor(mb * 1024 * 1024); + } let relatedSitesInput: unknown; try { @@ -162,6 +177,8 @@ export async function saveSiteAction( ...(siteUrl ? { siteUrl } : {}), ...(hubUrl ? { hubUrl } : {}), ...(pwa ? { pwa: true } : {}), + ...(archives ? {} : { archives: false }), + ...(archiveMaxBytes !== undefined ? { archiveMaxBytes } : {}), ...(relatedSites.length > 0 ? { relatedSites } : {}), }; try { diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx @@ -260,6 +260,32 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { this site installable with offline support. </p> + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + name="archives" + defaultChecked={initial.archives !== false} + className="accent-brand" + /> + Generate downloadable archive zips on build + </label> + <p className="-mt-2 text-xs text-muted-foreground"> + On by default: each build produces transcript &amp; live-chat zips (per + channel and combined) and lists them on the site&apos;s Downloads page. + Turn off to skip generation for this site. + </p> + <Field + label="Archive size cap (MB)" + name="archiveMaxMB" + type="number" + defaultValue={ + typeof initial.archiveMaxBytes === "number" + ? String(Math.round(initial.archiveMaxBytes / (1024 * 1024))) + : "" + } + hint="Archives larger than this are not served (they'd break a size-capped host like Cloudflare Pages' 25 MB limit) and show as unavailable. Leave blank for the default 25 MB; 0 = no cap." + /> + <fieldset className="flex flex-col gap-3 border border-border rounded p-3"> <legend className="px-1 text-sm font-medium">Channels</legend> <p className="text-xs text-muted-foreground"> @@ -603,21 +629,24 @@ function Field({ defaultValue, hint, required, + type = "text", }: { label: string; name: string; defaultValue?: string; hint?: string; required?: boolean; + type?: string; }) { return ( <label className="flex flex-col gap-1 text-sm"> <span className="font-medium">{label}</span> <input - type="text" + type={type} name={name} defaultValue={defaultValue} required={required} + min={type === "number" ? 0 : undefined} className="rounded border border-border bg-card px-2 py-1 text-sm" /> {hint && <span className="text-xs text-muted-foreground">{hint}</span>} diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,5 +1,9 @@ # Changelog +## [Unreleased] +- **A new Downloads page lets you take the whole archive with you.** Reachable from the header nav and footer (shown only when a build actually produced archives), `/downloads` lists the site's transcript and live-chat bundles as `.zip` downloads — the whole site up top, then per channel — each printing its video count and file size. A bundle too large to host (over the build's size cap) is shown as unavailable with the reason instead of a broken link. The zips are regenerated on every build. +- **Download a single video's transcript or live chat as a file.** The player toolbar has a new download control (`⤓`) that saves whatever you're viewing — the transcript, or the live chat — as `.txt`, `.srt`, or `.json`, generated in your browser from the already-loaded cues (no download of the full archive needed). + ## [0.6.0] - 2026-07-02 - **Kick videos now play in the site's own player.** Kick VODs have no embeddable player, so before this they could only link out to kick.com. The player now streams the VOD from its original HLS source directly — with the same real scrubbing, share-at-current-timestamp, and transcript-synced cue highlighting you get on YouTube, and better than the other embed-only platforms. Nothing is proxied through kick.com at view time. - **"Likely expired" badges for stream VODs that platforms delete.** Kick keeps VODs only ~30 days and Twitch keeps them 7–60 days (depending on the streamer's account), after which the video is gone and can't be played anywhere. Video cards for Kick/Twitch past that window now show a **Likely expired** badge (hover it for the platform's retention details). An expired Kick VOD shows a short "most likely deleted" notice with a link to the source instead of a broken player. The transcript stays fully searchable either way. diff --git a/export/app/components/Footer.tsx b/export/app/components/Footer.tsx @@ -9,6 +9,7 @@ import { } from "yt-dlp-transcript-common/lib/site"; import { currentSite } from "../lib/site"; import { instanceMode } from "../lib/mode"; +import { hasArchives } from "../lib/archives"; export default function Footer() { const site = currentSite(); @@ -32,6 +33,16 @@ export default function Footer() { > code.tar.gz </a> + {/* Transcript & live-chat archive zips, generated per build. Shown only + when this build actually produced servable archives. */} + {hasArchives() && ( + <a + href="/downloads" + className="underline underline-offset-2 hover:text-foreground transition-colors" + > + Transcripts &amp; chat + </a> + )} {/* The Offline (PWA download) entry point only shows on sites that opt into shipping an installable, offline-capable PWA (site.json `pwa: true`). Default-off keeps it hidden until configured. */} diff --git a/export/app/components/Header.tsx b/export/app/components/Header.tsx @@ -10,6 +10,7 @@ import { ThemeToggle } from "yt-dlp-transcript-common/components/ThemeToggle"; import { ThemeMenu } from "yt-dlp-transcript-common/components/ThemeMenu"; import { currentSite } from "../lib/site"; import { instanceMode } from "../lib/mode"; +import { hasArchives } from "../lib/archives"; import SiblingSwitcher from "./SiblingSwitcher"; // The export site's masthead: a brand-colored mark + wordmark, calm sans nav, @@ -33,6 +34,7 @@ export default function Header() { // The sibling switcher is build-time family navigation — it belongs on a // single site, not on the hub (whose "family" is the runtime shelf). const related = isSite ? resolveRelatedSites(site, listSites()) : []; + const showDownloads = hasArchives(); return ( <header className="sticky top-0 z-20 border-b border-border bg-background/80 backdrop-blur-md"> @@ -54,6 +56,14 @@ export default function Header() { > Duplicates </Link> + {showDownloads && ( + <Link + href="/downloads" + className="text-foreground hover:text-brand transition-colors" + > + Downloads + </Link> + )} </nav> <div className="ml-auto flex items-center gap-2 sm:gap-3 shrink-0"> diff --git a/export/app/downloads/page.tsx b/export/app/downloads/page.tsx @@ -0,0 +1,154 @@ +import type { Metadata } from "next"; +import type { ArchiveManifestEntry } from "yt-dlp-transcript-common/lib/archiveOptions"; +import { currentSite } from "../lib/site"; +import { readArchiveManifest } from "../lib/archives"; + +export const metadata: Metadata = { title: "Downloads" }; + +function humanBytes(n: number): string { + if (n <= 0) return "0 B"; + if (n < 1024) return `${n} B`; + const units = ["KB", "MB", "GB"]; + let v = n / 1024; + let i = 0; + while (v >= 1024 && i < units.length - 1) { + v /= 1024; + i++; + } + return `${v.toFixed(v >= 10 ? 0 : 1)} ${units[i]}`; +} + +function countLabel(kind: ArchiveManifestEntry["kind"], n: number): string { + const noun = kind === "transcripts" ? "transcript" : "chat log"; + return `${n.toLocaleString()} ${noun}${n === 1 ? "" : "s"}`; +} + +// One archive rendered as a manifest line: a printed contents label on the left, +// the download action on the right. Oversize entries were not shipped, so they +// read as unavailable with the reason rather than a dead link. +function ArchiveLine({ entry }: { entry: ArchiveManifestEntry }) { + const label = entry.channelTitle ?? (entry.scope === "all" ? "Everything" : entry.scope); + const kindLabel = entry.kind === "transcripts" ? "Transcripts" : "Live chat"; + + if (entry.oversize) { + return ( + <div className="flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-border/60 py-3 pl-4 opacity-70"> + <div className="min-w-0"> + <div className="font-medium text-foreground"> + {label} <span className="text-muted-foreground">· {kindLabel}</span> + </div> + <div className="font-mono text-xs text-muted-foreground"> + {countLabel(entry.kind, entry.videoCount)} · {humanBytes(entry.bytes)} + </div> + </div> + <span className="font-mono text-xs uppercase tracking-[0.12em] text-warning"> + Too large to host + </span> + </div> + ); + } + + return ( + <a + href={`/archives/${entry.filename}`} + download + className="group flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-brand/40 py-3 pl-4 transition-colors hover:border-brand hover:bg-muted/40" + > + <div className="min-w-0"> + <div className="font-medium text-foreground group-hover:text-brand transition-colors"> + {label} <span className="text-muted-foreground">· {kindLabel}</span> + </div> + <div className="font-mono text-xs text-muted-foreground"> + {countLabel(entry.kind, entry.videoCount)} · {entry.filename} + </div> + </div> + <span className="shrink-0 font-mono text-sm font-medium text-brand"> + ↓ {humanBytes(entry.bytes)} + </span> + </a> + ); +} + +export default function DownloadsPage() { + const site = currentSite(); + const manifest = readArchiveManifest(); + const entries = manifest?.entries ?? []; + + const combined = entries.filter((e) => e.scope === "all"); + const perChannel = entries.filter((e) => e.scope !== "all"); + const channelSlugs = Array.from(new Set(perChannel.map((e) => e.scope))); + const servedCount = entries.filter((e) => !e.oversize).length; + const totalBytes = entries + .filter((e) => !e.oversize) + .reduce((n, e) => n + e.bytes, 0); + + if (entries.length === 0) { + return ( + <div className="mx-auto flex max-w-2xl flex-col gap-3"> + <h1 className="font-display text-2xl font-semibold text-foreground"> + Downloads + </h1> + <p className="text-sm text-muted-foreground"> + No archives have been published for this site yet. Once a build with + archive generation runs, the transcript and live-chat download bundles + appear here. + </p> + </div> + ); + } + + return ( + <div className="mx-auto flex max-w-3xl flex-col gap-10"> + <header className="flex flex-col gap-3 border-b border-border pb-6"> + <p className="font-mono text-xs uppercase tracking-[0.18em] text-brand"> + Archive · {site.headerTitle} + </p> + <h1 className="font-display text-3xl font-semibold leading-tight text-foreground"> + Take the whole archive with you + </h1> + <p className="max-w-prose text-sm text-muted-foreground"> + Every transcript and live-chat log on this site, packaged as{" "} + <code className="font-mono">.zip</code> bundles. Each archive holds the + compact cues JSON per video plus a channel manifest — grab the whole + site or a single channel. + </p> + <p className="font-mono text-xs text-muted-foreground/80"> + {servedCount} bundle{servedCount === 1 ? "" : "s"} available ·{" "} + {humanBytes(totalBytes)} total + </p> + </header> + + {combined.length > 0 && ( + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + Whole site + </h2> + <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40"> + {combined.map((e) => ( + <ArchiveLine key={e.filename} entry={e} /> + ))} + </div> + </section> + )} + + {channelSlugs.length > 1 && ( + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + By channel + </h2> + <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40"> + {perChannel.map((e) => ( + <ArchiveLine key={e.filename} entry={e} /> + ))} + </div> + </section> + )} + + <p className="text-xs text-muted-foreground/70"> + Archives refresh on every site build. Transcripts and live chat are + packaged separately; channels with no captured live chat have no live-chat + bundle. + </p> + </div> + ); +} diff --git a/export/app/lib/archives.ts b/export/app/lib/archives.ts @@ -0,0 +1,30 @@ +import fs from "node:fs"; +import path from "node:path"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import type { ArchiveManifest } from "yt-dlp-transcript-common/lib/archiveOptions"; + +// The archive manifest compose-site.ts writes into public/archives at build +// time. Absent when the build didn't generate archives (dev, or opted out), so +// the Downloads page and its Header/Footer links all key off this presence. +// Read synchronously — this is a static export, evaluated once per build. +export function readArchiveManifest(): ArchiveManifest | null { + try { + const file = path.join( + getPaths().exportPublicDir, + "archives", + "manifest.json", + ); + const parsed = JSON.parse(fs.readFileSync(file, "utf8")) as ArchiveManifest; + if (!parsed || !Array.isArray(parsed.entries)) return null; + return parsed; + } catch { + return null; + } +} + +// Whether this site has at least one actually-served (non-oversize) archive to +// link to. Drives showing the Downloads nav/footer entry. +export function hasArchives(): boolean { + const manifest = readArchiveManifest(); + return !!manifest && manifest.entries.some((e) => !e.oversize); +}