commit 51a1b8f02c7dd7c5cea46020e8ef10164987d5b7
parent 71704bbfef276badb6308a505ddfeb925c775fe2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 3 Jul 2026 04:01:13 -0400
Add downloadable transcript & live-chat archive zips
Generate each site's transcript and live-chat archives during its build
into the served public/archives dir, expose them on a new Downloads page
(+ header/footer links), and add a per-video single-file download.
- zip is now the default archive format (was tar.gz); Build-page help copy
tracks the selected format.
- archive controllers take an optional outDir; compose-site builds per-channel
and combined zips scoped to the site's members, writes a size-aware
manifest.json, and drops any file over the cap (default 25 MB, configurable;
0 = no cap) so a Cloudflare Pages deploy isn't rejected.
- generation is on by default with three opt-out levels: global setting,
per-site site.json flag (+ archiveMaxBytes), and a per-build BUILD_ARCHIVES=0.
- Downloads page reads the manifest, listing whole-site and per-channel bundles
with sizes/counts; oversize entries render as unavailable.
- player toolbar gains a per-video download (.txt/.srt/.json) built client-side
from loaded cues, for whichever of transcript/live-chat is shown.
Verified: real compose-site run generates zips + manifest with correct
oversize handling; outDir produces a valid zip; cue-serializer unit tests pass;
common/editor/export all typecheck clean.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
26 files changed, 820 insertions(+), 25 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -60,6 +60,8 @@ yarn-error.log*
/export/public/_headers
/export/public/sw.js
/export/public/hub-sites.json
+# per-site bulk-download archive zips + manifest (regenerate with `pnpm build`)
+/export/public/archives/
/export/.export-index/
/transcripts/index.mdb/
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -14,16 +14,31 @@
// only that site's data. Checked-in static assets in public/ are left intact.
import path from "node:path";
-import { cp, mkdir, rm, readdir, access, readFile, writeFile } from "node:fs/promises";
+import { cp, mkdir, rm, readdir, access, readFile, writeFile, stat } from "node:fs/promises";
import { getPaths } from "../lib/paths";
import { getSite, resolveSocialLinks, resolveHubUrl, type Site } from "../lib/site";
+import { getSettings } from "../lib/settings";
import {
DUPLICATES_FILENAME,
filterClusterToChannels,
type DuplicateReport,
} from "../lib/duplicates";
-import type { Manifest } from "../lib/manifest";
+import type { Manifest, SubsManifest } from "../lib/manifest";
import { buildSiteDescriptor } from "../lib/siteDescriptor";
+import {
+ archiveTranscripts,
+ archiveCombinedTranscripts,
+} from "../controller/archiveTranscripts";
+import {
+ archiveLiveChat,
+ archiveCombinedLiveChat,
+} from "../controller/archiveLiveChat";
+import {
+ ARCHIVE_MANIFEST_FILENAME,
+ DEFAULT_ARCHIVE_MAX_BYTES,
+ type ArchiveManifest,
+ type ArchiveManifestEntry,
+} from "../lib/archiveOptions";
// CORS + cache headers for Cloudflare Pages (a static `_headers` file at the
// deploy root). All served data is public static JSON with no credentials, so
@@ -42,6 +57,8 @@ const CORS_HEADERS = `# Generated by compose-site.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/stats/*
Access-Control-Allow-Origin: *
+/archives/*
+ Access-Control-Allow-Origin: *
`;
// Emit the public federation contract: /site.json (branding + channels +
@@ -88,6 +105,196 @@ async function composeServiceWorker(
if (await exists(src)) await cp(src, dest);
}
+// Whether this build generates the downloadable archive zips. Opt-OUT at three
+// levels, all defaulting on: the global setting, the per-site `archives` flag,
+// and a per-build BUILD_ARCHIVES=0 env (set by the editor's "Skip archive zips"
+// control). Effective = AND of all three.
+function archivesEnabled(site: Site): boolean {
+ const globalOn = getSettings().buildArchives !== false;
+ const siteOn = site.archives !== false;
+ const buildOn = process.env.BUILD_ARCHIVES !== "0";
+ return globalOn && siteOn && buildOn;
+}
+
+// The served-file size cap for this site: per-site override, else MAX_ARCHIVE_BYTES
+// env, else the Cloudflare-safe default. 0 = no cap.
+function archiveMaxBytes(site: Site): number {
+ if (typeof site.archiveMaxBytes === "number" && site.archiveMaxBytes >= 0) {
+ return site.archiveMaxBytes;
+ }
+ const env = Number(process.env.MAX_ARCHIVE_BYTES);
+ if (Number.isFinite(env) && env >= 0) return env;
+ return DEFAULT_ARCHIVE_MAX_BYTES;
+}
+
+function humanBytes(n: number): string {
+ if (n < 1024) return `${n} B`;
+ const units = ["KB", "MB", "GB"];
+ let v = n / 1024;
+ let i = 0;
+ while (v >= 1024 && i < units.length - 1) {
+ v /= 1024;
+ i++;
+ }
+ return `${v.toFixed(v >= 10 || i === 0 ? 0 : 1)} ${units[i]}`;
+}
+
+// Channel display names keyed by slug, read from the already-composed per-site
+// subs manifest. Slugs are stable; display names are the human-facing titles.
+async function channelTitles(
+ paths: ReturnType<typeof getPaths>,
+): Promise<Map<string, string>> {
+ const titles = new Map<string, string>();
+ const manifestPath = path.join(paths.exportSubsDir, "manifest.json");
+ if (!(await exists(manifestPath))) return titles;
+ try {
+ const sm = JSON.parse(await readFile(manifestPath, "utf8")) as SubsManifest;
+ for (const c of sm.channels ?? []) {
+ if (c.slug && c.name) titles.set(c.slug, c.name);
+ }
+ } catch {
+ // A malformed subs manifest just means we fall back to slugs.
+ }
+ return titles;
+}
+
+// Stat a produced archive, and enforce the size cap: an oversize file is removed
+// from the served dir (so a capped host won't reject the whole deploy) and marked
+// oversize in the manifest, which the Downloads page renders as unavailable.
+async function describeArchive(
+ kind: ArchiveManifestEntry["kind"],
+ scope: string,
+ outDir: string,
+ filename: string,
+ videoCount: number,
+ maxBytes: number,
+ channelTitle?: string,
+): Promise<ArchiveManifestEntry> {
+ const filePath = path.join(outDir, filename);
+ let bytes = 0;
+ try {
+ bytes = (await stat(filePath)).size;
+ } catch {
+ bytes = 0;
+ }
+ const oversize = maxBytes > 0 && bytes > maxBytes;
+ if (oversize) {
+ await rm(filePath, { force: true });
+ console.warn(
+ `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — not served`,
+ );
+ }
+ const entry: ArchiveManifestEntry = { kind, scope, filename, bytes, videoCount };
+ if (channelTitle) entry.channelTitle = channelTitle;
+ if (oversize) entry.oversize = true;
+ return entry;
+}
+
+// Generate this site's bulk-download archive zips into public/archives and write
+// a manifest.json the Downloads page reads. Scoped to the site's member channels,
+// mirroring how every other per-site asset is composed. Runs on every build so
+// the zips always match the current corpus.
+async function composeArchives(
+ site: Site,
+ memberSlugs: string[],
+ paths: ReturnType<typeof getPaths>,
+): Promise<void> {
+ const outDir = path.join(paths.exportPublicDir, "archives");
+ // Always clear first so a disabled site (or one that lost the feature) never
+ // serves a stale bundle from a previous build.
+ await rm(outDir, { recursive: true, force: true });
+ if (!archivesEnabled(site)) {
+ console.log("[archives] disabled for this build — skipping.");
+ return;
+ }
+ await mkdir(outDir, { recursive: true });
+
+ const maxBytes = archiveMaxBytes(site);
+ const log = (m: string) => {
+ if (m) console.log(`[archives] ${m}`);
+ };
+ const build = { format: "zip" as const };
+ const common = { paths, channelSlugs: memberSlugs, outDir, build, onLog: log };
+
+ const transcripts = await archiveTranscripts(common);
+ const transcriptsAll = await archiveCombinedTranscripts(common);
+ const liveChat = await archiveLiveChat(common);
+ const liveChatAll = await archiveCombinedLiveChat(common);
+
+ const titles = await channelTitles(paths);
+ const bySlug = (a: { slug: string }, b: { slug: string }) =>
+ a.slug.localeCompare(b.slug);
+ const entries: ArchiveManifestEntry[] = [];
+
+ // Transcripts: combined ("all") first, then per-channel by slug.
+ if (transcriptsAll.archivePath) {
+ entries.push(
+ await describeArchive(
+ "transcripts",
+ "all",
+ outDir,
+ path.basename(transcriptsAll.archivePath),
+ transcriptsAll.totalTranscripts,
+ maxBytes,
+ ),
+ );
+ }
+ for (const a of [...transcripts.archives].sort(bySlug)) {
+ entries.push(
+ await describeArchive(
+ "transcripts",
+ a.slug,
+ outDir,
+ path.basename(a.archivePath),
+ a.transcriptCount,
+ maxBytes,
+ titles.get(a.slug),
+ ),
+ );
+ }
+
+ // Live chat: same ordering. Channels with no chat produce nothing to list.
+ if (liveChatAll.archivePath) {
+ entries.push(
+ await describeArchive(
+ "live-chat",
+ "all",
+ outDir,
+ path.basename(liveChatAll.archivePath),
+ liveChatAll.totalLiveChats,
+ maxBytes,
+ ),
+ );
+ }
+ for (const a of [...liveChat.archives].sort(bySlug)) {
+ entries.push(
+ await describeArchive(
+ "live-chat",
+ a.slug,
+ outDir,
+ path.basename(a.archivePath),
+ a.liveChatCount,
+ maxBytes,
+ titles.get(a.slug),
+ ),
+ );
+ }
+
+ const manifest: ArchiveManifest = {
+ version: 1,
+ thresholdBytes: maxBytes,
+ entries,
+ };
+ await writeFile(
+ path.join(outDir, ARCHIVE_MANIFEST_FILENAME),
+ JSON.stringify(manifest, null, 2) + "\n",
+ );
+ const served = entries.filter((e) => !e.oversize).length;
+ console.log(
+ `[archives] wrote ${served}/${entries.length} archive(s) to public/archives.`,
+ );
+}
+
async function exists(p: string): Promise<boolean> {
try {
await access(p);
@@ -201,6 +408,9 @@ async function main(): Promise<void> {
// --- service worker (only when this instance ships a PWA) ---
await composeServiceWorker(site, paths);
+ // --- bulk-download archive zips (on by default; see archivesEnabled) ---
+ await composeArchives(site, memberSlugs, paths);
+
const channelDirs = (await readdir(paths.exportTranscriptsDir).catch(
() => [] as string[],
)).length;
diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx
@@ -17,6 +17,7 @@ import { fetchSubs } from "./subsCache";
import { useUrlParams, writeUrlParams } from "./urlState";
import type { ModalMode } from "./urlState";
import { formatDate, formatDuration } from "../lib/format";
+import { cuesToSrt, cuesToText } from "../lib/vtt";
import type { Platform } from "../lib/transcripts";
import { vodExpiry } from "../lib/vodExpiry";
import { VodExpiredBadge } from "./badges";
@@ -125,6 +126,9 @@ type PlayerState = {
clearClip: () => void;
copyDownloadCommand: () => Promise<boolean>;
copyShareUrl: () => Promise<boolean>;
+ // Download the currently-shown transcript or live-chat log (per modalMode) as
+ // a single file, generated client-side from the loaded cues.
+ downloadTranscriptFile: (fmt: "json" | "txt" | "srt") => void;
openInPreservetube: () => void;
seekTo: (seconds: number) => void;
};
@@ -418,6 +422,53 @@ export function PlayerProvider({
);
}, [data]);
+ // Save the currently-shown cues (transcript or live chat, per modalMode) as a
+ // single file. Built entirely from cues already loaded in the browser — no
+ // server round-trip and no dependency on the bulk archive zips.
+ const downloadTranscriptFile = useCallback(
+ (fmt: "json" | "txt" | "srt") => {
+ const isChat = modalMode === "chat";
+ const cues = isChat
+ ? chat.slug === activeSlug
+ ? chat.cues
+ : null
+ : (data?.cues ?? null);
+ if (!data || !cues || cues.length === 0) return;
+
+ let body: string;
+ let mime: string;
+ if (fmt === "json") {
+ body = JSON.stringify(cues, null, 2);
+ mime = "application/json";
+ } else if (fmt === "srt") {
+ body = cuesToSrt(cues);
+ mime = "application/x-subrip";
+ } else {
+ body = cuesToText(cues);
+ mime = "text/plain";
+ }
+
+ const kind = isChat ? "live-chat" : "transcript";
+ const base =
+ (data.slug || data.title || kind)
+ .replace(/[^a-z0-9]+/gi, "-")
+ .replace(/^-+|-+$/g, "")
+ .toLowerCase() || kind;
+ const filename = `${base}.${kind}.${fmt}`;
+
+ const blob = new Blob([body], { type: `${mime};charset=utf-8` });
+ const url = URL.createObjectURL(blob);
+ const a = document.createElement("a");
+ a.href = url;
+ a.download = filename;
+ document.body.appendChild(a);
+ a.click();
+ a.remove();
+ URL.revokeObjectURL(url);
+ },
+ [data, chat, activeSlug, modalMode],
+ );
+
const seekTo = useCallback((seconds: number) => {
if (playerRef.current) {
playerRef.current.seekTo(seconds, "seconds");
@@ -614,6 +665,7 @@ export function PlayerProvider({
clearClip,
copyDownloadCommand,
copyShareUrl,
+ downloadTranscriptFile,
openInPreservetube,
seekTo,
}),
@@ -636,6 +688,7 @@ export function PlayerProvider({
clearClip,
copyDownloadCommand,
copyShareUrl,
+ downloadTranscriptFile,
openInPreservetube,
seekTo,
],
diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx
@@ -41,6 +41,7 @@ export default function TranscriptModal() {
clearClip,
copyDownloadCommand,
copyShareUrl,
+ downloadTranscriptFile,
openInPreservetube,
seekTo,
} = usePlayer();
@@ -50,6 +51,8 @@ export default function TranscriptModal() {
const copyResetRef = useRef<number | null>(null);
const [shareCopied, setShareCopied] = useState(false);
const shareResetRef = useRef<number | null>(null);
+ const [downloadOpen, setDownloadOpen] = useState(false);
+ const downloadRef = useRef<HTMLDivElement | null>(null);
// Tracks whether the next scrollToIndex should animate. Smooth on
// user-initiated changes (cue click, mode toggle, initial open); instant
// on natural playhead drift so 4 Hz progress ticks don't keep retriggering
@@ -57,6 +60,31 @@ export default function TranscriptModal() {
const scrollKindRef = useRef<"smooth" | "auto">("smooth");
const isChat = modalMode === "chat";
+ const canDownloadFile = isChat
+ ? (chatCues?.length ?? 0) > 0
+ : (data?.cues?.length ?? 0) > 0;
+
+ // Close the download format menu on an outside click or Escape.
+ useEffect(() => {
+ if (!downloadOpen) return;
+ const onDown = (e: MouseEvent) => {
+ if (
+ downloadRef.current &&
+ !downloadRef.current.contains(e.target as Node)
+ ) {
+ setDownloadOpen(false);
+ }
+ };
+ const onKey = (e: KeyboardEvent) => {
+ if (e.key === "Escape") setDownloadOpen(false);
+ };
+ document.addEventListener("mousedown", onDown);
+ document.addEventListener("keydown", onKey);
+ return () => {
+ document.removeEventListener("mousedown", onDown);
+ document.removeEventListener("keydown", onKey);
+ };
+ }, [downloadOpen]);
// Pre-compute author/body split once per cue array so the render hot path
// doesn't redo `indexOf`/`slice` on every progress tick. Memo key is the
@@ -225,6 +253,36 @@ export default function TranscriptModal() {
char={shareCopied ? "✓" : "⤴"}
disabled={!data}
/>
+ <div className="relative" ref={downloadRef}>
+ <ControlButton
+ title={
+ canDownloadFile
+ ? `Download this ${isChat ? "live chat" : "transcript"} as a file`
+ : "Nothing to download yet"
+ }
+ onClick={() => setDownloadOpen((v) => !v)}
+ char="⤓"
+ disabled={!canDownloadFile}
+ highlight={downloadOpen}
+ />
+ {downloadOpen && (
+ <div className="absolute right-0 top-9 z-10 flex flex-col overflow-hidden rounded-md bg-zinc-900 shadow-xl ring-1 ring-white/15">
+ {(["txt", "srt", "json"] as const).map((fmt) => (
+ <button
+ key={fmt}
+ type="button"
+ onClick={() => {
+ downloadTranscriptFile(fmt);
+ setDownloadOpen(false);
+ }}
+ className="px-3 py-1.5 text-left font-mono text-xs uppercase tracking-wide text-zinc-200 hover:bg-white/10"
+ >
+ .{fmt}
+ </button>
+ ))}
+ </div>
+ )}
+ </div>
{data?.platform === "youtube" && (
<ControlButton
title="Open in Preservetube"
diff --git a/common/controller/archiveLiveChat.ts b/common/controller/archiveLiveChat.ts
@@ -46,6 +46,10 @@ export type ArchiveLiveChatOptions = {
concurrency?: number;
channelSlugs?: string[];
build?: Partial<ArchiveBuildOptions>;
+ // Destination for the finished archives. Defaults to transcripts/export/
+ // archives; compose-site passes the served public/archives dir. Mirrors
+ // ArchiveTranscriptsOptions.outDir.
+ outDir?: string;
};
export type ArchiveLiveChatResult = {
@@ -185,7 +189,8 @@ export async function archiveLiveChat(
`Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`,
);
- await mkdir(archivesDir(opts.paths), { recursive: true });
+ const outDir = opts.outDir ?? archivesDir(opts.paths);
+ await mkdir(outDir, { recursive: true });
await mkdir(stagingDir(opts.paths), { recursive: true });
const allChannels = await listChannels(opts.paths);
@@ -200,7 +205,7 @@ export async function archiveLiveChat(
}
const staging = path.join(stagingDir(opts.paths), ch.slug);
const archivePath = path.join(
- archivesDir(opts.paths),
+ outDir,
`${ch.slug}.${ARCHIVE_INFIX}.${archiveExtension(build.format)}`,
);
try {
@@ -244,7 +249,7 @@ export async function archiveLiveChat(
log(
`Wrote ${archives.length} archive${archives.length === 1 ? "" : "s"} containing ${totalLiveChats} live chats to:`,
);
- log(` ${archivesDir(opts.paths)}`);
+ log(` ${outDir}`);
for (const a of archives) {
log(` - ${path.basename(a.archivePath)} (${a.liveChatCount} live chats)`);
}
@@ -271,7 +276,8 @@ export async function archiveCombinedLiveChat(
`Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`,
);
- await mkdir(archivesDir(opts.paths), { recursive: true });
+ const outDir = opts.outDir ?? archivesDir(opts.paths);
+ await mkdir(outDir, { recursive: true });
await mkdir(stagingDir(opts.paths), { recursive: true });
const combinedStaging = path.join(
@@ -279,7 +285,7 @@ export async function archiveCombinedLiveChat(
COMBINED_STAGING_DIR,
);
const archivePath = path.join(
- archivesDir(opts.paths),
+ outDir,
`${COMBINED_BASENAME}.${archiveExtension(build.format)}`,
);
diff --git a/common/controller/archiveTranscripts.ts b/common/controller/archiveTranscripts.ts
@@ -60,6 +60,10 @@ export type ArchiveTranscriptsOptions = {
concurrency?: number;
channelSlugs?: string[];
build?: Partial<ArchiveBuildOptions>;
+ // Destination for the finished archives. Defaults to transcripts/export/
+ // archives (the standalone editor actions' location); compose-site passes the
+ // served public/archives dir so the zips ship with the site build.
+ outDir?: string;
};
export type ArchiveTranscriptsResult = {
@@ -281,7 +285,8 @@ export async function archiveTranscripts(
`Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`,
);
- await mkdir(archivesDir(opts.paths), { recursive: true });
+ const outDir = opts.outDir ?? archivesDir(opts.paths);
+ await mkdir(outDir, { recursive: true });
await mkdir(stagingDir(opts.paths), { recursive: true });
const allChannels = await listChannels(opts.paths);
@@ -296,7 +301,7 @@ export async function archiveTranscripts(
}
const staging = path.join(stagingDir(opts.paths), ch.slug);
const archivePath = path.join(
- archivesDir(opts.paths),
+ outDir,
`${ch.slug}.${archiveExtension(build.format)}`,
);
try {
@@ -340,7 +345,7 @@ export async function archiveTranscripts(
log(
`Wrote ${archives.length} archive${archives.length === 1 ? "" : "s"} containing ${totalTranscripts} transcripts to:`,
);
- log(` ${archivesDir(opts.paths)}`);
+ log(` ${outDir}`);
for (const a of archives) {
log(` - ${path.basename(a.archivePath)} (${a.transcriptCount} transcripts)`);
}
@@ -367,12 +372,13 @@ export async function archiveCombinedTranscripts(
`Format: ${build.format} level=${build.compressionLevel} metadata=${build.includeMetadata} pretty=${build.prettyPrint}`,
);
- await mkdir(archivesDir(opts.paths), { recursive: true });
+ const outDir = opts.outDir ?? archivesDir(opts.paths);
+ await mkdir(outDir, { recursive: true });
await mkdir(stagingDir(opts.paths), { recursive: true });
const combinedStaging = path.join(stagingDir(opts.paths), COMBINED_STAGING_DIR);
const archivePath = path.join(
- archivesDir(opts.paths),
+ outDir,
`${COMBINED_BASENAME}.${archiveExtension(build.format)}`,
);
diff --git a/common/lib/archiveOptions.ts b/common/lib/archiveOptions.ts
@@ -5,10 +5,11 @@
export type ArchiveFormat = "tar.gz" | "tar.xz" | "zip";
+// `zip` first so it leads the editor's format dropdown, matching the default.
export const ARCHIVE_FORMATS: ReadonlyArray<ArchiveFormat> = [
+ "zip",
"tar.gz",
"tar.xz",
- "zip",
];
export type ArchiveBuildOptions = {
@@ -24,8 +25,42 @@ export type ArchiveBuildOptions = {
};
export const DEFAULT_ARCHIVE_OPTIONS: ArchiveBuildOptions = {
- format: "tar.gz",
+ format: "zip",
compressionLevel: 6,
includeMetadata: true,
prettyPrint: false,
};
+
+// --- Served archive manifest -------------------------------------------------
+// compose-site.ts writes public/archives/manifest.json describing the bulk
+// download zips it produced for a site. The Downloads page reads it to list the
+// archives with sizes/counts without hardcoding filenames. Client-safe (no fs).
+
+export type ArchiveManifestEntry = {
+ kind: "transcripts" | "live-chat";
+ // "all" for the combined archive, otherwise a channel slug.
+ scope: "all" | string;
+ // Present when scope is a channel slug.
+ channelTitle?: string;
+ // Filename relative to /archives/ (e.g. "all-transcripts.zip", "grumps.zip").
+ filename: string;
+ bytes: number;
+ videoCount: number;
+ // True when the file exceeded thresholdBytes: it is NOT served (removed so a
+ // size-capped host like Cloudflare Pages won't reject the whole deploy) and
+ // the UI renders it as unavailable rather than a broken link.
+ oversize?: boolean;
+};
+
+export type ArchiveManifest = {
+ version: 1;
+ // The size cap applied this build, so the UI can explain omissions. 0 = no cap.
+ thresholdBytes: number;
+ entries: ArchiveManifestEntry[];
+};
+
+export const ARCHIVE_MANIFEST_FILENAME = "manifest.json";
+
+// Default per-file size cap. Cloudflare Pages rejects any single asset > 25 MB,
+// which is why oversize combined archives must be dropped from what's served.
+export const DEFAULT_ARCHIVE_MAX_BYTES = 25 * 1024 * 1024;
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -97,6 +97,11 @@ export type SiteSettings = {
// video once the stream ends. Per-channel override available
// (ChannelConfig.skipLiveDownloads).
skipLiveDownloads: boolean;
+ // Whether site builds generate downloadable transcript/live-chat archive zips
+ // (into public/archives, linked on the Downloads page). Global default; a site
+ // can opt out via site.json `archives: false`, and a single build can skip via
+ // the "Skip archive zips" build control. Opt-out: default true.
+ buildArchives: boolean;
// Debounce preset for the global snapshot scheduler: how long it waits after
// the last report-changing action before regenerating affected channel
// reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap).
@@ -449,6 +454,7 @@ function defaults(): SiteSettings {
parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
inlineTranscribeOnFallback: false,
skipLiveDownloads: true,
+ buildArchives: true,
reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET,
autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS,
syncScheduler: defaultSyncScheduler(),
@@ -637,6 +643,9 @@ export function getSettings(): SiteSettings {
if (typeof merged.skipLiveDownloads !== "boolean") {
merged.skipLiveDownloads = true;
}
+ if (typeof merged.buildArchives !== "boolean") {
+ merged.buildArchives = true;
+ }
if (!isReportDebouncePreset(merged.reportDebouncePreset)) {
merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET;
}
@@ -813,6 +822,7 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
),
inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true,
skipLiveDownloads: next.skipLiveDownloads !== false,
+ buildArchives: next.buildArchives !== false,
reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset)
? next.reportDebouncePreset
: DEFAULT_REPORT_DEBOUNCE_PRESET,
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -76,6 +76,17 @@ export type Site = {
// trusts only the hub PWA. Set true to make this site its own installable PWA.
// See export/app/lib/mode.ts (shipsPwa) and common/lib/siteDescriptor.ts.
pwa?: boolean;
+ // Whether the site build generates downloadable transcript/live-chat archive
+ // zips into public/archives (and links them on the Downloads page). This is
+ // an opt-OUT: undefined/true = on, only explicit `false` disables. Also gated
+ // by the global setting and a per-build flag (see compose-site.ts). Default-on
+ // because bulk download is the point of publishing a corpus.
+ archives?: boolean;
+ // Per-site override for the served-file size cap (bytes). Any archive larger
+ // than this is dropped from what's served and flagged in the manifest so a
+ // capped host (Cloudflare Pages: 25 MB) won't reject the deploy. 0 = no cap.
+ // Absent = the global DEFAULT_ARCHIVE_MAX_BYTES / MAX_ARCHIVE_BYTES env.
+ archiveMaxBytes?: number;
// Per-site override for the hub this site belongs under (the PWA it points
// visitors toward). Absent = inherit the family default SiteSettings.homepageUrl.
// Resolve with resolveHubUrl(). Surfaced on /site.json so a hub can tell member
@@ -246,6 +257,14 @@ export function parseSite(siteId: string, raw: unknown): Site {
siteUrl: parseSiteUrl(r.siteUrl),
relatedSites: parseRelatedSites(r.relatedSites),
pwa: r.pwa === true,
+ // Opt-out: only an explicit false disables. Absent/true stays on.
+ archives: r.archives !== false,
+ archiveMaxBytes:
+ typeof r.archiveMaxBytes === "number" &&
+ Number.isFinite(r.archiveMaxBytes) &&
+ r.archiveMaxBytes >= 0
+ ? Math.floor(r.archiveMaxBytes)
+ : undefined,
hubUrl: parseSiteUrl(r.hubUrl),
};
}
@@ -408,6 +427,13 @@ export async function writeSite(
? { relatedSites: parseRelatedSites(site.relatedSites) }
: {}),
...(site.pwa ? { pwa: true } : {}),
+ // Persist only the non-default: archives is on unless explicitly disabled.
+ ...(site.archives === false ? { archives: false } : {}),
+ ...(typeof site.archiveMaxBytes === "number" &&
+ Number.isFinite(site.archiveMaxBytes) &&
+ site.archiveMaxBytes >= 0
+ ? { archiveMaxBytes: Math.floor(site.archiveMaxBytes) }
+ : {}),
...(parseSiteUrl(site.hubUrl) ? { hubUrl: parseSiteUrl(site.hubUrl) } : {}),
};
const dir = siteDir(paths, site.siteId);
diff --git a/common/lib/vtt.test.ts b/common/lib/vtt.test.ts
@@ -0,0 +1,39 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { cuesToText, cuesToSrt, type Cue } from "./vtt";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/vtt.test.ts
+
+const cues: Cue[] = [
+ { start: 0, end: 2.5, text: "Hello there" },
+ { start: 3661.25, end: 3663, text: "General Kenobi" },
+];
+
+test("cuesToText writes one line per cue, trailing newline", () => {
+ assert.equal(cuesToText(cues), "Hello there\nGeneral Kenobi\n");
+});
+
+test("cuesToText on empty cues is a lone newline", () => {
+ assert.equal(cuesToText([]), "\n");
+});
+
+test("cuesToSrt numbers blocks and formats HH:MM:SS,mmm timing", () => {
+ const srt = cuesToSrt(cues);
+ assert.equal(
+ srt,
+ "1\n00:00:00,000 --> 00:00:02,500\nHello there\n" +
+ "\n" +
+ "2\n01:01:01,250 --> 01:01:03,000\nGeneral Kenobi\n",
+ );
+});
+
+test("cuesToSrt falls back to start when end is missing", () => {
+ const noEnd = [{ start: 5, end: undefined as unknown as number, text: "x" }];
+ const srt = cuesToSrt(noEnd);
+ assert.equal(srt, "1\n00:00:05,000 --> 00:00:05,000\nx\n");
+});
+
+test("cuesToSrt clamps negative times to zero", () => {
+ const neg = [{ start: -3, end: -1, text: "neg" }];
+ assert.equal(cuesToSrt(neg), "1\n00:00:00,000 --> 00:00:00,000\nneg\n");
+});
diff --git a/common/lib/vtt.ts b/common/lib/vtt.ts
@@ -74,6 +74,34 @@ export function parseVtt(src: string): Cue[] {
return deduped;
}
+// One text line per cue. For live chat each cue's text is already "author:
+// message", so this doubles as a readable chat log.
+export function cuesToText(cues: Cue[]): string {
+ return cues.map((c) => c.text).join("\n") + "\n";
+}
+
+function srtTime(seconds: number): string {
+ const s = Math.max(0, seconds);
+ const h = Math.floor(s / 3600);
+ const m = Math.floor((s % 3600) / 60);
+ const sec = Math.floor(s % 60);
+ const ms = Math.round((s - Math.floor(s)) * 1000);
+ const pad = (n: number, w = 2) => String(n).padStart(w, "0");
+ return `${pad(h)}:${pad(m)}:${pad(sec)},${pad(ms, 3)}`;
+}
+
+// SubRip (.srt): numbered, timestamped cue blocks. `end` falls back to `start`
+// so zero-length cues (some chat lines) still produce valid timing.
+export function cuesToSrt(cues: Cue[]): string {
+ return cues
+ .map((c, i) => {
+ const start = srtTime(c.start);
+ const end = srtTime(c.end ?? c.start);
+ return `${i + 1}\n${start} --> ${end}\n${c.text}\n`;
+ })
+ .join("\n");
+}
+
export function formatTimestamp(seconds: number): string {
const s = Math.floor(seconds);
const h = Math.floor(s / 3600);
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Every site build now bundles downloadable transcript & live-chat archive zips.** The archive builders (per-channel `<slug>.zip`, combined `all-transcripts.zip` / `all-live-chat.zip`) previously only ran as standalone actions that wrote to a non-served directory; now `compose-site` generates them for the site's own channels straight into the served `public/archives/` and writes a `manifest.json` (sizes + counts) that the site's new **Downloads** page reads. **`zip` is now the default archive format** everywhere (was `tar.gz`), and the Build page's format help text tracks the selected format. Generation is **on by default with three opt-out levels**: a global **Generate archive zips on build** toggle in Settings, a per-site **Generate archive zips** toggle (plus an optional **Archive size cap (MB)**) on the site's page, and a per-build **Skip archive zips** checkbox on the Build and Build & Deploy controls (`BUILD_ARCHIVES=0`). Because a single file over ~25 MB breaks a Cloudflare Pages deploy, any archive over the cap (default 25 MB; `0` = no cap) is dropped from what's served and flagged `oversize` in the manifest so the deploy still succeeds and the Downloads page shows it as unavailable rather than a dead link. See `common/bin/compose-site.ts` (`composeArchives`), `common/controller/archive{Transcripts,LiveChat}.ts` (new `outDir` option), `common/lib/archiveOptions.ts` (default + manifest types), `common/lib/{site,settings}.ts` (opt-out flags), and `editor/app/{deploy/buildDeployCore.ts,build/buildAction.ts,deploy/components/Build{Export,Deploy}Button.tsx,sites/components/SiteForm.tsx,settings/components/SettingsForm.tsx}`.
- **You can now change a channel's slug (its id) — deliberately, from the Danger zone.** A channel's slug *is* its on-disk directory name (`transcripts/channels/<slug>/`), so it used to be fixed at creation ("Slug is fixed once a channel is created"). A new **Rename** form in the channel's Danger zone lifts that: enter a new slug and **type the current slug to confirm** (same friction as delete), and the rename is blocked while the channel has running/queued jobs (the in-memory registry keys by slug). Because the slug is a directory name, the rename does a **full migration** of every slug-keyed store so nothing silently breaks: it moves the channel dir (config, data, playlist, snapshot, shards, failed lists) **and** the saved-video store dir — rewriting each `saved-video.json` pointer's absolute `dir` so persisted source videos still resolve — then retargets every site.json membership, the sync scheduler's per-channel backoff state, and any job bookmarks. The two filesystem moves run first and roll back on failure; the metadata updates that follow are atomic and best-effort (surfaced as warnings). Renaming **changes the channel's public URL** (the old one 404s), which the form warns about. The slug grammar is also now validated on create. See `common/controller/renameChannel.ts`, `common/controller/channels.ts` (`isValidChannelSlug`), `common/lib/savedVideo-server.ts` (`rewriteSavedVideoDir`), `common/jobs/bookmarks.ts` (`renameChannelInBookmarks`), `editor/app/channels/{actions.ts,components/RenameChannelForm.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-rename.spec.ts`.
- **New Queue diagnostics page (`/jobs/queue`): see & force-release stuck jobs.** The job system has two sources of truth that can drift — the registry owns each job's `status`, the scheduler owns the running SLOT per queue. A cancel that never finalizes (a child that ignored SIGTERM, a crashed finalizer) leaves a job "cancelled" in the registry while the scheduler still marks its slot running, silently blocking every job behind it on that queue — and the Active Jobs page hides it (it filters to running/queued). The new **Queue** page reconciles the two: it builds from the **scheduler** as the source of truth for slots, cross-checks each against its registry record, and flags a running head as **stuck** when the record is terminal-but-holding-slot, evicted, or (softer) a live job idle past 10 minutes. It **auto-heals** the hard cases on every view/poll (frees terminal/evicted slots), shows a health strip (active queues, running, queued, **stuck**, workers), per-queue cards with the held-for duration / PID (`kill -9` hint) / last log line, and a **Force-release** button per slot (SIGKILLs the child and frees the slot unconditionally) plus a **Reap all stuck** action. Force-release is also available on any running job in Active Jobs, and Active Jobs links to Queue with a stuck-count badge. See `common/jobs/registry.ts` (`forceRelease`), `editor/app/jobs/queue/*`, `editor/app/jobs/{actions.ts,components/ForceReleaseJobButton.tsx}`, and `editor/e2e/queue.spec.ts`.
- **Jobs page: real log retention + pagination (replaces the dead "Clear archived logs" button).** The old button only deleted logs absent from the in-memory registry — which, since the registry keeps the 100 newest finished jobs and sidecars preserve their real status, was almost never anything, so it did nothing. It's replaced by a **Clear logs** dropdown that prunes finished-job logs by age (older than 7 / 30 / 90 days) or all at once; running/queued jobs are never deleted. The `.jobs` directory also **self-trims on job finish** (throttled; keep newest 500, drop >30 days) so it can't grow unbounded. Job ids are now **ULIDs** (lexicographically time-sortable, timestamp decodable from the id), letting the list **paginate** — `listAllJobs` returns one page (default 50, grown by a **Load more** link) and only `stat`s/reads the sidecar for the shown page instead of every file on every load. `jobIdTime()` decodes both ULID and the legacy `<t36>-<rand>` ids, so existing on-disk logs still sort/read correctly. See `common/jobs/{ulid,listJobs,registry,streamCommand}.ts`, `editor/app/jobs/{page.tsx,actions.ts,components/ClearLogsMenu.tsx,[id]/page.tsx}`, and `editor/e2e/jobs.spec.ts`.
diff --git a/editor/app/build/buildAction.ts b/editor/app/build/buildAction.ts
@@ -64,6 +64,7 @@ export async function buildExportAction(
siteId: string,
queueKey?: string,
skipData?: boolean,
+ skipArchives?: boolean,
): Promise<StreamActionResult> {
const paths = getPaths();
const id = siteId.trim();
@@ -81,7 +82,10 @@ export async function buildExportAction(
queueKey: queueKey === undefined ? DEFAULT_BUILD_QUEUE : queueKey.trim(),
paths,
fn: async (onLog, signal) => {
- const code = await runBuildPhase(onLog, signal, id, paths, { skipData });
+ const code = await runBuildPhase(onLog, signal, id, paths, {
+ skipData,
+ skipArchives,
+ });
if (signal.aborted) return;
if (code !== 0) throw new Error(`Build failed (exit ${code})`);
},
@@ -94,6 +98,7 @@ export async function buildExportAction(
// deploy queue so it never overlaps a standalone deploy or another build-deploy.
export async function buildAndDeployAction(
siteId: string,
+ skipArchives?: boolean,
): Promise<StreamActionResult> {
const paths = getPaths();
const id = siteId.trim();
@@ -113,7 +118,9 @@ export async function buildAndDeployAction(
paths,
fn: async (onLog, signal) => {
onLog("=== Build ===\n");
- const buildCode = await runBuildPhase(onLog, signal, id, paths);
+ const buildCode = await runBuildPhase(onLog, signal, id, paths, {
+ skipArchives,
+ });
// A cancel mid-build must NOT proceed to deploy.
if (signal.aborted) return;
if (buildCode !== 0) {
diff --git a/editor/app/build/components/BuildButtons.tsx b/editor/app/build/components/BuildButtons.tsx
@@ -140,7 +140,7 @@ export function BuildButtons({ existingQueues }: Props) {
<div>
<h2 className="text-lg font-semibold">Build transcript archives</h2>
<p className="text-sm text-muted-foreground">
- Produces one <code><channel>.tar.gz</code> per channel
+ Produces one <code><channel>.{archiveOptions.format}</code> per channel
under <code>transcripts/export/archives/</code>, containing each
video's compact <code>transcript.cues.json</code> plus a{" "}
<code>channel.json</code> manifest. Normalizes missing cues on
@@ -178,7 +178,7 @@ export function BuildButtons({ existingQueues }: Props) {
</h2>
<p className="text-sm text-muted-foreground">
Produces a single{" "}
- <code>all-transcripts.tar.gz</code> under{" "}
+ <code>all-transcripts.{combinedOptions.format}</code> under{" "}
<code>transcripts/export/archives/</code> containing every
channel plus a top-level <code>manifest.json</code> index.
Normalizes missing cues on demand.
@@ -241,7 +241,7 @@ export function BuildButtons({ existingQueues }: Props) {
<div>
<h2 className="text-lg font-semibold">Build live chat archives</h2>
<p className="text-sm text-muted-foreground">
- Produces one <code><channel>.live_chat.tar.gz</code> per
+ Produces one <code><channel>.live_chat.{liveChatArchiveOptions.format}</code> per
channel under <code>transcripts/export/archives/</code>,
containing each video's parsed{" "}
<code>live_chat.cues.json</code> plus a{" "}
@@ -281,7 +281,7 @@ export function BuildButtons({ existingQueues }: Props) {
Build combined live chat archive
</h2>
<p className="text-sm text-muted-foreground">
- Produces a single <code>all-live-chat.tar.gz</code> under{" "}
+ Produces a single <code>all-live-chat.{liveChatCombinedOptions.format}</code> under{" "}
<code>transcripts/export/archives/</code> containing every
channel's parsed live chats plus a top-level{" "}
<code>manifest.json</code> index. Normalizes missing live chats
diff --git a/editor/app/deploy/buildDeployCore.ts b/editor/app/deploy/buildDeployCore.ts
@@ -30,7 +30,7 @@ export async function runBuildPhase(
signal: AbortSignal,
siteId: string,
paths: Paths,
- opts?: { skipData?: boolean },
+ opts?: { skipData?: boolean; skipArchives?: boolean },
): Promise<number> {
const { mode } = getSettings().buildPipeline;
if (mode === "docker") {
@@ -51,6 +51,12 @@ export async function runBuildPhase(
"existing .export-index staging.\n",
);
}
+ // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes
+ // compose-site skip generation this build regardless of the global/site flags.
+ const skipArchives = opts?.skipArchives === true;
+ if (skipArchives) {
+ onLog("[notice] Skipping archive-zip generation for this build.\n");
+ }
return runChildIntoLog(onLog, signal, {
command: "pnpm",
args: ["run", skipData ? "build:nodata" : "build"],
@@ -61,6 +67,7 @@ export async function runBuildPhase(
TRANSCRIPTS_DIR: paths.transcriptsDir,
EXPORT_PUBLIC_DIR: paths.exportPublicDir,
SITE_ID: siteId,
+ ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}),
},
});
}
diff --git a/editor/app/deploy/components/BuildDeployButton.tsx b/editor/app/deploy/components/BuildDeployButton.tsx
@@ -1,5 +1,6 @@
"use client";
+import { useState } from "react";
import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog";
import { buildAndDeployAction } from "../../build/buildAction";
import { cancelJobAction } from "../../jobs/actions";
@@ -20,6 +21,7 @@ export function BuildDeployButton({
cloudflareProject,
}: Props) {
const ready = Boolean(siteId && cloudflareProject);
+ const [skipArchives, setSkipArchives] = useState(false);
let notice: React.ReactNode;
if (!siteId) {
@@ -46,13 +48,27 @@ export function BuildDeployButton({
return (
<StreamActionLog
- trigger={() => buildAndDeployAction(siteId as string)}
+ trigger={() => buildAndDeployAction(siteId as string, skipArchives)}
cancelAction={cancelJobAction}
buttonLabel="Build & deploy"
runningLabel="Building & deploying…"
label="Build and deploy"
disabled={!ready}
- extraControls={notice}
+ extraControls={
+ <div className="flex flex-col gap-2">
+ {notice}
+ {ready && (
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ checked={skipArchives}
+ onChange={(e) => setSkipArchives(e.target.checked)}
+ />
+ Skip archive zips (transcript & live-chat downloads)
+ </label>
+ )}
+ </div>
+ }
/>
);
}
diff --git a/editor/app/deploy/components/BuildExportButton.tsx b/editor/app/deploy/components/BuildExportButton.tsx
@@ -18,10 +18,13 @@ const DEFAULT_BUILD_QUEUE = "build";
export function BuildExportButton({ existingQueues, siteId, siteTitle }: Props) {
const [exportQueue, setExportQueue] = useState(DEFAULT_BUILD_QUEUE);
const [skipData, setSkipData] = useState(false);
+ const [skipArchives, setSkipArchives] = useState(false);
return (
<StreamActionLog
- trigger={() => buildExportAction(siteId as string, exportQueue, skipData)}
+ trigger={() =>
+ buildExportAction(siteId as string, exportQueue, skipData, skipArchives)
+ }
cancelAction={cancelJobAction}
buttonLabel="Build static export"
runningLabel="Building static export…"
@@ -50,6 +53,18 @@ export function BuildExportButton({ existingQueues, siteId, siteTitle }: Props)
Reuses the existing index/stats — only composes the site and runs
next build. Do a full build first if data is stale.
</p>
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ checked={skipArchives}
+ onChange={(e) => setSkipArchives(e.target.checked)}
+ />
+ Skip archive zips (transcript & live-chat downloads)
+ </label>
+ <p className="text-xs text-muted-foreground">
+ Skips generating the downloadable archive zips for this build only.
+ Leave off to keep the site's Downloads page current.
+ </p>
<QueueControl
value={exportQueue}
onChange={setExportQueue}
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -50,6 +50,7 @@ export async function saveSettingsAction(
const inlineTranscribeOnFallback =
formData.get("inlineTranscribeOnFallback") === "on";
const skipLiveDownloads = formData.get("skipLiveDownloads") === "on";
+ const buildArchives = formData.get("buildArchives") === "on";
const reportDebouncePresetRaw = String(
formData.get("reportDebouncePreset") ?? "",
).trim();
@@ -212,6 +213,7 @@ export async function saveSettingsAction(
parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
inlineTranscribeOnFallback,
skipLiveDownloads,
+ buildArchives,
reportDebouncePreset,
autoRefreshIntervalSeconds: autoRefreshParsed,
syncScheduler,
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -163,6 +163,25 @@ export function SettingsForm({ initial, apps }: Props) {
</span>
</span>
</label>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="buildArchives"
+ defaultChecked={initial.buildArchives}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Generate downloadable archive zips on build
+ </span>
+ <span className="text-xs text-muted-foreground">
+ Each site build produces transcript & live-chat archive zips (per
+ channel and combined) and lists them on the site's Downloads
+ page. On by default. A site can opt out on its own page, and a single
+ build can skip them from the Build controls.
+ </span>
+ </span>
+ </label>
<label className="flex flex-col gap-1 text-sm">
<span className="font-medium">Report refresh debounce</span>
<select
diff --git a/editor/app/sites/actions.ts b/editor/app/sites/actions.ts
@@ -71,6 +71,21 @@ export async function saveSiteAction(
}
const hubUrl = parseSiteUrl(hubUrlRaw);
const pwa = formData.get("pwa") === "on";
+ // Archives default on: an unchecked (default-on) box yields no "archives" key
+ // → false → persisted as the explicit opt-out.
+ const archives = formData.get("archives") === "on";
+ const archiveMaxMBRaw = String(formData.get("archiveMaxMB") ?? "").trim();
+ let archiveMaxBytes: number | undefined;
+ if (archiveMaxMBRaw) {
+ const mb = Number(archiveMaxMBRaw);
+ if (!Number.isFinite(mb) || mb < 0) {
+ return {
+ ok: false,
+ error: "Archive size cap must be a non-negative number of MB, or blank.",
+ };
+ }
+ archiveMaxBytes = Math.floor(mb * 1024 * 1024);
+ }
let relatedSitesInput: unknown;
try {
@@ -162,6 +177,8 @@ export async function saveSiteAction(
...(siteUrl ? { siteUrl } : {}),
...(hubUrl ? { hubUrl } : {}),
...(pwa ? { pwa: true } : {}),
+ ...(archives ? {} : { archives: false }),
+ ...(archiveMaxBytes !== undefined ? { archiveMaxBytes } : {}),
...(relatedSites.length > 0 ? { relatedSites } : {}),
};
try {
diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx
@@ -260,6 +260,32 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
this site installable with offline support.
</p>
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="archives"
+ defaultChecked={initial.archives !== false}
+ className="accent-brand"
+ />
+ Generate downloadable archive zips on build
+ </label>
+ <p className="-mt-2 text-xs text-muted-foreground">
+ On by default: each build produces transcript & live-chat zips (per
+ channel and combined) and lists them on the site's Downloads page.
+ Turn off to skip generation for this site.
+ </p>
+ <Field
+ label="Archive size cap (MB)"
+ name="archiveMaxMB"
+ type="number"
+ defaultValue={
+ typeof initial.archiveMaxBytes === "number"
+ ? String(Math.round(initial.archiveMaxBytes / (1024 * 1024)))
+ : ""
+ }
+ hint="Archives larger than this are not served (they'd break a size-capped host like Cloudflare Pages' 25 MB limit) and show as unavailable. Leave blank for the default 25 MB; 0 = no cap."
+ />
+
<fieldset className="flex flex-col gap-3 border border-border rounded p-3">
<legend className="px-1 text-sm font-medium">Channels</legend>
<p className="text-xs text-muted-foreground">
@@ -603,21 +629,24 @@ function Field({
defaultValue,
hint,
required,
+ type = "text",
}: {
label: string;
name: string;
defaultValue?: string;
hint?: string;
required?: boolean;
+ type?: string;
}) {
return (
<label className="flex flex-col gap-1 text-sm">
<span className="font-medium">{label}</span>
<input
- type="text"
+ type={type}
name={name}
defaultValue={defaultValue}
required={required}
+ min={type === "number" ? 0 : undefined}
className="rounded border border-border bg-card px-2 py-1 text-sm"
/>
{hint && <span className="text-xs text-muted-foreground">{hint}</span>}
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,9 @@
# Changelog
+## [Unreleased]
+- **A new Downloads page lets you take the whole archive with you.** Reachable from the header nav and footer (shown only when a build actually produced archives), `/downloads` lists the site's transcript and live-chat bundles as `.zip` downloads — the whole site up top, then per channel — each printing its video count and file size. A bundle too large to host (over the build's size cap) is shown as unavailable with the reason instead of a broken link. The zips are regenerated on every build.
+- **Download a single video's transcript or live chat as a file.** The player toolbar has a new download control (`⤓`) that saves whatever you're viewing — the transcript, or the live chat — as `.txt`, `.srt`, or `.json`, generated in your browser from the already-loaded cues (no download of the full archive needed).
+
## [0.6.0] - 2026-07-02
- **Kick videos now play in the site's own player.** Kick VODs have no embeddable player, so before this they could only link out to kick.com. The player now streams the VOD from its original HLS source directly — with the same real scrubbing, share-at-current-timestamp, and transcript-synced cue highlighting you get on YouTube, and better than the other embed-only platforms. Nothing is proxied through kick.com at view time.
- **"Likely expired" badges for stream VODs that platforms delete.** Kick keeps VODs only ~30 days and Twitch keeps them 7–60 days (depending on the streamer's account), after which the video is gone and can't be played anywhere. Video cards for Kick/Twitch past that window now show a **Likely expired** badge (hover it for the platform's retention details). An expired Kick VOD shows a short "most likely deleted" notice with a link to the source instead of a broken player. The transcript stays fully searchable either way.
diff --git a/export/app/components/Footer.tsx b/export/app/components/Footer.tsx
@@ -9,6 +9,7 @@ import {
} from "yt-dlp-transcript-common/lib/site";
import { currentSite } from "../lib/site";
import { instanceMode } from "../lib/mode";
+import { hasArchives } from "../lib/archives";
export default function Footer() {
const site = currentSite();
@@ -32,6 +33,16 @@ export default function Footer() {
>
code.tar.gz
</a>
+ {/* Transcript & live-chat archive zips, generated per build. Shown only
+ when this build actually produced servable archives. */}
+ {hasArchives() && (
+ <a
+ href="/downloads"
+ className="underline underline-offset-2 hover:text-foreground transition-colors"
+ >
+ Transcripts & chat
+ </a>
+ )}
{/* The Offline (PWA download) entry point only shows on sites that
opt into shipping an installable, offline-capable PWA (site.json
`pwa: true`). Default-off keeps it hidden until configured. */}
diff --git a/export/app/components/Header.tsx b/export/app/components/Header.tsx
@@ -10,6 +10,7 @@ import { ThemeToggle } from "yt-dlp-transcript-common/components/ThemeToggle";
import { ThemeMenu } from "yt-dlp-transcript-common/components/ThemeMenu";
import { currentSite } from "../lib/site";
import { instanceMode } from "../lib/mode";
+import { hasArchives } from "../lib/archives";
import SiblingSwitcher from "./SiblingSwitcher";
// The export site's masthead: a brand-colored mark + wordmark, calm sans nav,
@@ -33,6 +34,7 @@ export default function Header() {
// The sibling switcher is build-time family navigation — it belongs on a
// single site, not on the hub (whose "family" is the runtime shelf).
const related = isSite ? resolveRelatedSites(site, listSites()) : [];
+ const showDownloads = hasArchives();
return (
<header className="sticky top-0 z-20 border-b border-border bg-background/80 backdrop-blur-md">
@@ -54,6 +56,14 @@ export default function Header() {
>
Duplicates
</Link>
+ {showDownloads && (
+ <Link
+ href="/downloads"
+ className="text-foreground hover:text-brand transition-colors"
+ >
+ Downloads
+ </Link>
+ )}
</nav>
<div className="ml-auto flex items-center gap-2 sm:gap-3 shrink-0">
diff --git a/export/app/downloads/page.tsx b/export/app/downloads/page.tsx
@@ -0,0 +1,154 @@
+import type { Metadata } from "next";
+import type { ArchiveManifestEntry } from "yt-dlp-transcript-common/lib/archiveOptions";
+import { currentSite } from "../lib/site";
+import { readArchiveManifest } from "../lib/archives";
+
+export const metadata: Metadata = { title: "Downloads" };
+
+function humanBytes(n: number): string {
+ if (n <= 0) return "0 B";
+ if (n < 1024) return `${n} B`;
+ const units = ["KB", "MB", "GB"];
+ let v = n / 1024;
+ let i = 0;
+ while (v >= 1024 && i < units.length - 1) {
+ v /= 1024;
+ i++;
+ }
+ return `${v.toFixed(v >= 10 ? 0 : 1)} ${units[i]}`;
+}
+
+function countLabel(kind: ArchiveManifestEntry["kind"], n: number): string {
+ const noun = kind === "transcripts" ? "transcript" : "chat log";
+ return `${n.toLocaleString()} ${noun}${n === 1 ? "" : "s"}`;
+}
+
+// One archive rendered as a manifest line: a printed contents label on the left,
+// the download action on the right. Oversize entries were not shipped, so they
+// read as unavailable with the reason rather than a dead link.
+function ArchiveLine({ entry }: { entry: ArchiveManifestEntry }) {
+ const label = entry.channelTitle ?? (entry.scope === "all" ? "Everything" : entry.scope);
+ const kindLabel = entry.kind === "transcripts" ? "Transcripts" : "Live chat";
+
+ if (entry.oversize) {
+ return (
+ <div className="flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-border/60 py-3 pl-4 opacity-70">
+ <div className="min-w-0">
+ <div className="font-medium text-foreground">
+ {label} <span className="text-muted-foreground">· {kindLabel}</span>
+ </div>
+ <div className="font-mono text-xs text-muted-foreground">
+ {countLabel(entry.kind, entry.videoCount)} · {humanBytes(entry.bytes)}
+ </div>
+ </div>
+ <span className="font-mono text-xs uppercase tracking-[0.12em] text-warning">
+ Too large to host
+ </span>
+ </div>
+ );
+ }
+
+ return (
+ <a
+ href={`/archives/${entry.filename}`}
+ download
+ className="group flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-brand/40 py-3 pl-4 transition-colors hover:border-brand hover:bg-muted/40"
+ >
+ <div className="min-w-0">
+ <div className="font-medium text-foreground group-hover:text-brand transition-colors">
+ {label} <span className="text-muted-foreground">· {kindLabel}</span>
+ </div>
+ <div className="font-mono text-xs text-muted-foreground">
+ {countLabel(entry.kind, entry.videoCount)} · {entry.filename}
+ </div>
+ </div>
+ <span className="shrink-0 font-mono text-sm font-medium text-brand">
+ ↓ {humanBytes(entry.bytes)}
+ </span>
+ </a>
+ );
+}
+
+export default function DownloadsPage() {
+ const site = currentSite();
+ const manifest = readArchiveManifest();
+ const entries = manifest?.entries ?? [];
+
+ const combined = entries.filter((e) => e.scope === "all");
+ const perChannel = entries.filter((e) => e.scope !== "all");
+ const channelSlugs = Array.from(new Set(perChannel.map((e) => e.scope)));
+ const servedCount = entries.filter((e) => !e.oversize).length;
+ const totalBytes = entries
+ .filter((e) => !e.oversize)
+ .reduce((n, e) => n + e.bytes, 0);
+
+ if (entries.length === 0) {
+ return (
+ <div className="mx-auto flex max-w-2xl flex-col gap-3">
+ <h1 className="font-display text-2xl font-semibold text-foreground">
+ Downloads
+ </h1>
+ <p className="text-sm text-muted-foreground">
+ No archives have been published for this site yet. Once a build with
+ archive generation runs, the transcript and live-chat download bundles
+ appear here.
+ </p>
+ </div>
+ );
+ }
+
+ return (
+ <div className="mx-auto flex max-w-3xl flex-col gap-10">
+ <header className="flex flex-col gap-3 border-b border-border pb-6">
+ <p className="font-mono text-xs uppercase tracking-[0.18em] text-brand">
+ Archive · {site.headerTitle}
+ </p>
+ <h1 className="font-display text-3xl font-semibold leading-tight text-foreground">
+ Take the whole archive with you
+ </h1>
+ <p className="max-w-prose text-sm text-muted-foreground">
+ Every transcript and live-chat log on this site, packaged as{" "}
+ <code className="font-mono">.zip</code> bundles. Each archive holds the
+ compact cues JSON per video plus a channel manifest — grab the whole
+ site or a single channel.
+ </p>
+ <p className="font-mono text-xs text-muted-foreground/80">
+ {servedCount} bundle{servedCount === 1 ? "" : "s"} available ·{" "}
+ {humanBytes(totalBytes)} total
+ </p>
+ </header>
+
+ {combined.length > 0 && (
+ <section className="flex flex-col gap-3">
+ <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground">
+ Whole site
+ </h2>
+ <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40">
+ {combined.map((e) => (
+ <ArchiveLine key={e.filename} entry={e} />
+ ))}
+ </div>
+ </section>
+ )}
+
+ {channelSlugs.length > 1 && (
+ <section className="flex flex-col gap-3">
+ <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground">
+ By channel
+ </h2>
+ <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40">
+ {perChannel.map((e) => (
+ <ArchiveLine key={e.filename} entry={e} />
+ ))}
+ </div>
+ </section>
+ )}
+
+ <p className="text-xs text-muted-foreground/70">
+ Archives refresh on every site build. Transcripts and live chat are
+ packaged separately; channels with no captured live chat have no live-chat
+ bundle.
+ </p>
+ </div>
+ );
+}
diff --git a/export/app/lib/archives.ts b/export/app/lib/archives.ts
@@ -0,0 +1,30 @@
+import fs from "node:fs";
+import path from "node:path";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import type { ArchiveManifest } from "yt-dlp-transcript-common/lib/archiveOptions";
+
+// The archive manifest compose-site.ts writes into public/archives at build
+// time. Absent when the build didn't generate archives (dev, or opted out), so
+// the Downloads page and its Header/Footer links all key off this presence.
+// Read synchronously — this is a static export, evaluated once per build.
+export function readArchiveManifest(): ArchiveManifest | null {
+ try {
+ const file = path.join(
+ getPaths().exportPublicDir,
+ "archives",
+ "manifest.json",
+ );
+ const parsed = JSON.parse(fs.readFileSync(file, "utf8")) as ArchiveManifest;
+ if (!parsed || !Array.isArray(parsed.entries)) return null;
+ return parsed;
+ } catch {
+ return null;
+ }
+}
+
+// Whether this site has at least one actually-served (non-oversize) archive to
+// link to. Drives showing the Downloads nav/footer entry.
+export function hasArchives(): boolean {
+ const manifest = readArchiveManifest();
+ return !!manifest && manifest.entries.some((e) => !e.oversize);
+}