commit 75e3e74f16b5bc5f7ad653662f114b728ac76ed5
parent 55d1fbc5e705118e62515fe3b4690b3194adc3ee
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 7 Jul 2026 01:36:57 -0400
Incremental export build: skip unchanged channels
Archives and every compose-site copy used to redo all work each build.
Gate them on a cheap per-channel content signature so unchanged channels
are skipped — a no-change recompose drops from re-zipping/re-copying
gigabytes to a few seconds, with byte-identical output.
- common/lib/channelSignature.ts: per-channel signature from the mtimes
LMDB sub-DB (build:index's freshness truth; shard-safe, unlike raw
fs.stat). Signs inputs, not the (timestamp-bearing) output bytes.
- archiveTranscripts.ts/archiveLiveChat.ts: build each channel's zip once
into the persistent shared cache (<slug>.zip + <slug>.sig sidecar),
reused build-to-build and across sites; parallelize the channel loop.
- compose-site.ts: build archives into the shared cache then hard-link the
member subset into public/; replace rm -rf + full cp of the transcript/
subs/summaries/stats trees with an in-place reconcile gated on a
dirSignature of the source (manifest.json excluded — its generatedAt
churns every build).
- buildDeployCore.ts: skip R2 re-upload of an oversize archive when R2
already holds a same-size object (HeadObject), since cache-hit zips are
byte-stable.
- gitignore /export/.compose-cache/.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 630 insertions(+), 81 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -70,6 +70,8 @@ yarn-error.log*
# oversize archives staged for the deploy-time R2 upload
/export/.r2-staging/
/export/.export-index/
+# per-site incremental-compose signatures (skip-unchanged cache)
+/export/.compose-cache/
/transcripts/index.mdb/
# homepage (hub) generated public data (regenerate with `pnpm build:homepage`)
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -14,7 +14,9 @@
// only that site's data. Checked-in static assets in public/ are left intact.
import path from "node:path";
-import { cp, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises";
+import { cp, link, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises";
+import { createHash } from "node:crypto";
+import type { Dirent } from "node:fs";
import { getPaths } from "../lib/paths";
import { getSite, resolveSocialLinks, resolveHubUrl, type Site } from "../lib/site";
import { getSettings } from "../lib/settings";
@@ -32,8 +34,12 @@ import {
renderRobotsTxt,
renderSitemapXml,
} from "../lib/corpus";
-import { archiveTranscripts } from "../controller/archiveTranscripts";
+import {
+ archiveCacheDir,
+ archiveTranscripts,
+} from "../controller/archiveTranscripts";
import { archiveLiveChat } from "../controller/archiveLiveChat";
+import { openChannelSigner } from "../lib/channelSignature";
import {
ARCHIVE_MANIFEST_FILENAME,
DEFAULT_ARCHIVE_MAX_BYTES,
@@ -332,13 +338,49 @@ async function composeArchives(
if (m) console.log(`[archives] ${m}`);
};
const build = { format: "zip" as const };
- const common = { paths, channelSlugs: memberSlugs, outDir, build, onLog: log };
- // Per-channel archives only — the combined "whole-site" bundles were dropped
- // (they duplicated the per-channel content and were always the first to blow
- // past the size cap).
- const transcripts = await archiveTranscripts(common);
- const liveChat = await archiveLiveChat(common);
+ // Build (or reuse) archives in the persistent SHARED cache — not straight into
+ // public/ — so a channel unchanged since the last build (or already built for
+ // another site this run) is not re-zipped. One signer view is shared across the
+ // transcripts + live-chat passes. compose then copies this site's member subset
+ // out of the cache into public/archives below.
+ const cacheDir = archiveCacheDir(paths);
+ const signer = openChannelSigner(paths);
+ let transcripts: Awaited<ReturnType<typeof archiveTranscripts>>;
+ let liveChat: Awaited<ReturnType<typeof archiveLiveChat>>;
+ try {
+ const common = {
+ paths,
+ channelSlugs: memberSlugs,
+ outDir: cacheDir,
+ build,
+ onLog: log,
+ signer,
+ };
+ // Per-channel archives only — the combined "whole-site" bundles were dropped
+ // (they duplicated the per-channel content and were always the first to blow
+ // past the size cap).
+ transcripts = await archiveTranscripts(common);
+ liveChat = await archiveLiveChat(common);
+ } finally {
+ await signer.close();
+ }
+
+ // Materialize each cached archive into this site's served tree. Hard-link when
+ // possible (same filesystem, read-only served asset) so it's near-free; fall
+ // back to a byte copy across devices. describeArchive() below then stats the
+ // public copy and applies the size cap / R2 overflow exactly as before,
+ // leaving the shared cache original untouched.
+ const materialize = async (srcPath: string): Promise<void> => {
+ const dest = path.join(outDir, path.basename(srcPath));
+ try {
+ await link(srcPath, dest);
+ } catch {
+ await cp(srcPath, dest);
+ }
+ };
+ for (const a of transcripts.archives) await materialize(a.archivePath);
+ for (const a of liveChat.archives) await materialize(a.archivePath);
const titles = await channelTitles(paths);
const bySlug = (a: { slug: string }, b: { slug: string }) =>
@@ -410,6 +452,151 @@ async function replaceDir(src: string, dest: string): Promise<void> {
}
}
+// --- Incremental compose cache ------------------------------------------------
+// Per-site record of what we last materialized into public/, keyed by a cheap
+// content signature of each source. When the signature is unchanged and the
+// destination is still present, the copy/serialize is skipped. Kept outside
+// public/ (a sibling of .export-index / .r2-staging) so it's never deployed.
+type ComposeCache = {
+ transcripts: Record<string, string>;
+ subs: Record<string, string>;
+ summaries?: string;
+ stats?: string;
+ duplicates?: string;
+};
+
+function composeCachePath(
+ paths: ReturnType<typeof getPaths>,
+ siteId: string,
+): string {
+ return path.join(
+ path.dirname(paths.exportPublicDir),
+ ".compose-cache",
+ `${siteId}.json`,
+ );
+}
+
+async function readComposeCache(p: string): Promise<ComposeCache> {
+ try {
+ const parsed = JSON.parse(await readFile(p, "utf8"));
+ return {
+ transcripts: parsed?.transcripts ?? {},
+ subs: parsed?.subs ?? {},
+ summaries: parsed?.summaries,
+ stats: parsed?.stats,
+ duplicates: parsed?.duplicates,
+ };
+ } catch {
+ return { transcripts: {}, subs: {} };
+ }
+}
+
+async function writeComposeCache(p: string, cache: ComposeCache): Promise<void> {
+ await mkdir(path.dirname(p), { recursive: true });
+ await writeFile(p, JSON.stringify(cache));
+}
+
+// A cheap content signature for a directory: relative path + size + mtime of
+// every file within, hashed. It reflects exactly what a recursive copy would
+// move, so an unchanged source (build:index skipped it incrementally) yields the
+// same signature and the copy is skipped. The shared transcript/subs trees hold
+// only a bounded handful of paginated page files per channel, so this is far
+// cheaper than the copy it guards. Returns "" when the dir is absent/empty.
+//
+// `ignoreBasename` excludes files by name from the signature. The shared
+// per-channel trees carry a manifest.json that build:index rewrites with a fresh
+// `generatedAt` for EVERY channel whenever ANY channel mutates (the shared
+// page-writer loop runs over all channels). Its pageCount/slugToPage only change
+// when the channel's pages change — which the page files already capture — so
+// excluding it lets an unchanged channel stay skipped on a partial-change build
+// instead of re-copying all 50-odd channels. The tiny stale manifest left behind
+// on a skip is structurally identical (only its timestamp differs).
+async function dirSignature(
+ dir: string,
+ ignoreBasename?: string,
+): Promise<string> {
+ const h = createHash("sha1");
+ let any = false;
+ const walk = async (rel: string): Promise<void> => {
+ let ents: Dirent[];
+ try {
+ ents = await readdir(path.join(dir, rel), { withFileTypes: true });
+ } catch {
+ return;
+ }
+ ents.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
+ for (const e of ents) {
+ if (!e.isDirectory() && e.name === ignoreBasename) continue;
+ const childRel = rel ? `${rel}/${e.name}` : e.name;
+ if (e.isDirectory()) {
+ await walk(childRel);
+ } else {
+ any = true;
+ const s = await stat(path.join(dir, childRel));
+ h.update(`${childRel}\t${s.size}\t${s.mtimeMs}\n`);
+ }
+ }
+ };
+ await walk("");
+ return any ? h.digest("hex") : "";
+}
+
+// Materialize the member subset of a shared per-channel tree into public/,
+// skipping channels whose source is unchanged since the last compose. Replaces
+// the previous "rm -rf the whole tree then cp every member" with an in-place
+// reconcile: only changed channels are re-copied, and channels no longer members
+// are pruned. Returns the new per-slug signature map for the compose cache.
+async function reconcileChannelTree(
+ kind: string,
+ srcRoot: string,
+ destRoot: string,
+ memberSlugs: string[],
+ prev: Record<string, string>,
+ log: (m: string) => void,
+): Promise<Record<string, string>> {
+ await mkdir(destRoot, { recursive: true });
+ const next: Record<string, string> = {};
+ const memberSet = new Set(memberSlugs);
+ let copied = 0;
+ let skipped = 0;
+ for (const slug of memberSlugs) {
+ const src = path.join(srcRoot, slug);
+ const dest = path.join(destRoot, slug);
+ // Exclude the per-channel manifest.json — its `generatedAt` churns every
+ // mutation build; the page files capture real content changes.
+ const sig = await dirSignature(src, "manifest.json");
+ if (sig === "") {
+ // No source for this member — ensure no stale dest survives.
+ await rm(dest, { recursive: true, force: true });
+ continue;
+ }
+ if (prev[slug] === sig && (await exists(dest))) {
+ next[slug] = sig;
+ skipped++;
+ continue;
+ }
+ await rm(dest, { recursive: true, force: true });
+ await cp(src, dest, { recursive: true });
+ next[slug] = sig;
+ copied++;
+ }
+ // Prune channel dirs that are no longer members (leaves non-dir siblings such
+ // as the per-site subs manifest.json untouched).
+ let entries: Dirent[];
+ try {
+ entries = await readdir(destRoot, { withFileTypes: true });
+ } catch {
+ entries = [];
+ }
+ for (const e of entries) {
+ if (e.isDirectory() && !memberSet.has(e.name)) {
+ await rm(path.join(destRoot, e.name), { recursive: true, force: true });
+ }
+ }
+ log(`[compose] ${kind}: ${copied} copied, ${skipped} unchanged.`);
+ return next;
+}
+
async function main(): Promise<void> {
const siteId = process.env.SITE_ID;
if (!siteId) {
@@ -420,38 +607,50 @@ async function main(): Promise<void> {
const site = getSite(siteId, paths);
const memberSlugs = site.channels.map((c) => c.slug);
- // --- per-site aggregates (whole-dir / single-file swaps) ---
- await replaceDir(
- path.join(paths.exportSitesIndexDir, siteId, "summaries"),
- paths.exportSummariesDir,
- );
- await replaceDir(
- path.join(paths.exportSitesIndexDir, siteId, "stats"),
- paths.exportStatsDir,
- );
-
- // --- shared per-channel trees, filtered to this site's members ---
- // Transcripts: replace the whole tree with the member subset.
- await rm(paths.exportTranscriptsDir, { recursive: true, force: true });
- await mkdir(paths.exportTranscriptsDir, { recursive: true });
- for (const slug of memberSlugs) {
- const src = path.join(paths.exportSharedTranscriptsDir, slug);
- if (await exists(src)) {
- await cp(src, path.join(paths.exportTranscriptsDir, slug), {
- recursive: true,
- });
- }
+ // Incremental compose: skip stages whose source is unchanged since last build.
+ const cachePath = composeCachePath(paths, siteId);
+ const cache = await readComposeCache(cachePath);
+
+ // --- per-site aggregates (whole-dir swaps), gated on the source signature ---
+ const summariesSrc = path.join(paths.exportSitesIndexDir, siteId, "summaries");
+ const summariesSig = await dirSignature(summariesSrc);
+ if (
+ cache.summaries !== summariesSig ||
+ !(await exists(paths.exportSummariesDir))
+ ) {
+ await replaceDir(summariesSrc, paths.exportSummariesDir);
+ cache.summaries = summariesSig;
+ } else {
+ console.log("[compose] summaries: unchanged.");
}
-
- // Subs: per-channel dirs are shared; the site-level manifest.json is per-site.
- await rm(paths.exportSubsDir, { recursive: true, force: true });
- await mkdir(paths.exportSubsDir, { recursive: true });
- for (const slug of memberSlugs) {
- const src = path.join(paths.exportSharedSubsDir, slug);
- if (await exists(src)) {
- await cp(src, path.join(paths.exportSubsDir, slug), { recursive: true });
- }
+ const statsSrc = path.join(paths.exportSitesIndexDir, siteId, "stats");
+ const statsSig = await dirSignature(statsSrc);
+ if (cache.stats !== statsSig || !(await exists(paths.exportStatsDir))) {
+ await replaceDir(statsSrc, paths.exportStatsDir);
+ cache.stats = statsSig;
+ } else {
+ console.log("[compose] stats: unchanged.");
}
+
+ // --- shared per-channel trees, filtered to this site's members ---
+ // Reconcile the member subset in place (only changed channels are re-copied).
+ cache.transcripts = await reconcileChannelTree(
+ "transcripts",
+ paths.exportSharedTranscriptsDir,
+ paths.exportTranscriptsDir,
+ memberSlugs,
+ cache.transcripts,
+ console.log,
+ );
+ cache.subs = await reconcileChannelTree(
+ "subs",
+ paths.exportSharedSubsDir,
+ paths.exportSubsDir,
+ memberSlugs,
+ cache.subs,
+ console.log,
+ );
+ // Subs also carries a per-site manifest.json (tiny — copied every build).
const subsManifestSrc = path.join(
paths.exportSitesIndexDir,
siteId,
@@ -494,25 +693,54 @@ async function main(): Promise<void> {
// nav link. No file → the page shows its empty state and the link self-hides.
const dupSrc = path.join(paths.transcriptsDir, DUPLICATES_FILENAME);
const dupDest = path.join(paths.exportPublicDir, DUPLICATES_FILENAME);
- await rm(dupDest, { force: true });
- if (site.duplicates !== false && (await exists(dupSrc))) {
- const report = JSON.parse(await readFile(dupSrc, "utf8")) as DuplicateReport;
- const memberSet = new Set(memberSlugs);
- const clusters = report.clusters
- .map((c) => filterClusterToChannels(c, memberSet))
- .filter((c): c is NonNullable<typeof c> => c !== null);
- if (clusters.length > 0) {
- const filtered: DuplicateReport = {
- ...report,
- totals: {
- ...report.totals,
- clusters: clusters.length,
- videosInClusters: clusters.reduce((n, c) => n + c.videoRefs.length, 0),
- },
- clusters,
- };
- await writeFile(dupDest, JSON.stringify(filtered));
+ // Gate the re-filter/re-serialize on the source's mtime+size, the member set,
+ // and the feature flag. The cache value is prefixed written|/empty| so a skip
+ // can self-heal if public/ was wiped out of band (presence must match).
+ let dupStat: { mtimeMs: number; size: number } | null = null;
+ try {
+ const s = await stat(dupSrc);
+ dupStat = { mtimeMs: s.mtimeMs, size: s.size };
+ } catch {
+ dupStat = null;
+ }
+ const dupEnabled = site.duplicates !== false && dupStat !== null;
+ const dupKey = `${dupEnabled}|${dupStat?.mtimeMs ?? ""}|${dupStat?.size ?? ""}|${[...memberSlugs].sort().join(",")}`;
+ const dupDestPresent = await exists(dupDest);
+ const dupCachedWritten = cache.duplicates?.startsWith("written|") ?? false;
+ const dupCacheHit =
+ cache.duplicates?.endsWith(`|${dupKey}`) &&
+ dupCachedWritten === dupDestPresent;
+ if (dupCacheHit) {
+ console.log("[compose] duplicates: unchanged.");
+ } else {
+ await rm(dupDest, { force: true });
+ let wrote = false;
+ if (dupEnabled) {
+ const report = JSON.parse(
+ await readFile(dupSrc, "utf8"),
+ ) as DuplicateReport;
+ const memberSet = new Set(memberSlugs);
+ const clusters = report.clusters
+ .map((c) => filterClusterToChannels(c, memberSet))
+ .filter((c): c is NonNullable<typeof c> => c !== null);
+ if (clusters.length > 0) {
+ const filtered: DuplicateReport = {
+ ...report,
+ totals: {
+ ...report.totals,
+ clusters: clusters.length,
+ videosInClusters: clusters.reduce(
+ (n, c) => n + c.videoRefs.length,
+ 0,
+ ),
+ },
+ clusters,
+ };
+ await writeFile(dupDest, JSON.stringify(filtered));
+ wrote = true;
+ }
}
+ cache.duplicates = `${wrote ? "written" : "empty"}|${dupKey}`;
}
// --- federation contract: /site.json descriptor + CORS _headers ---
@@ -528,6 +756,9 @@ async function main(): Promise<void> {
// (after site.json + archives — both feed into these fixed-count files)
await emitAiFiles(paths);
+ // Persist the incremental-compose signatures for the next build.
+ await writeComposeCache(cachePath, cache);
+
const channelDirs = (await readdir(paths.exportTranscriptsDir).catch(
() => [] as string[],
)).length;
diff --git a/common/controller/archiveLiveChat.ts b/common/controller/archiveLiveChat.ts
@@ -16,6 +16,7 @@
import path from "node:path";
import {
+ access,
link,
mkdir,
readdir,
@@ -28,10 +29,15 @@ import { listChannels, type ChannelStat } from "./channels";
import { normalizeLiveChat } from "./normalizeLiveChat";
import {
archiveExtension,
+ archiveSidecarPath,
+ archiveSignatureExtra,
compressArchive,
+ readArchiveSidecar,
resolveBuild,
+ writeArchiveSidecar,
writeTransformedCues,
} from "./archiveTranscripts";
+import { openChannelSigner, type ChannelSigner } from "../lib/channelSignature";
import { LIVE_CHAT_CUES_FILENAME } from "../lib/videoStatus";
import { detectPlatform } from "../lib/platform";
import type { Paths } from "../lib/paths";
@@ -39,17 +45,31 @@ import {
type ArchiveBuildOptions,
} from "../lib/archiveOptions";
+async function fileExists(p: string): Promise<boolean> {
+ try {
+ await access(p);
+ return true;
+ } catch {
+ return false;
+ }
+}
+
export type ArchiveLiveChatOptions = {
paths: Paths;
onLog?: (msg: string) => void;
signal?: AbortSignal;
concurrency?: number;
+ // Channels built concurrently (see ArchiveTranscriptsOptions.channelConcurrency).
+ channelConcurrency?: number;
channelSlugs?: string[];
build?: Partial<ArchiveBuildOptions>;
// Destination for the finished archives. Defaults to transcripts/export/
// archives; compose-site passes the served public/archives dir. Mirrors
// ArchiveTranscriptsOptions.outDir.
outDir?: string;
+ // Reuse an already-open per-channel signer across passes (see
+ // ArchiveTranscriptsOptions.signer).
+ signer?: ChannelSigner;
};
export type ArchiveLiveChatResult = {
@@ -198,16 +218,40 @@ export async function archiveLiveChat(
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
- for (const ch of target) {
- if (opts.signal?.aborted) {
- log("Aborted.");
- break;
- }
- const staging = path.join(stagingDir(opts.paths), ch.slug);
+ const signer = opts.signer ?? openChannelSigner(opts.paths);
+ const channelLimit = pLimit(opts.channelConcurrency ?? 4);
+ let reused = 0;
+
+ const buildOne = async (
+ ch: ChannelStat,
+ ): Promise<{
+ archive?: ArchiveLiveChatResult["archives"][number];
+ skipped?: ArchiveLiveChatResult["skipped"][number];
+ reused?: boolean;
+ }> => {
+ if (opts.signal?.aborted) return {};
const archivePath = path.join(
outDir,
`${ch.slug}.${ARCHIVE_INFIX}.${archiveExtension(build.format)}`,
);
+ // `live-chat:` marks this signature as the live-chat variant so it never
+ // collides with the transcripts archive's sidecar (different files anyway,
+ // but keeps the two kinds' signatures independent).
+ const sig = signer.signature(
+ ch.slug,
+ `live-chat:${archiveSignatureExtra(build, ch)}`,
+ );
+ if (sig) {
+ const prev = await readArchiveSidecar(archivePath);
+ if (prev && prev.signature === sig && (await fileExists(archivePath))) {
+ log(` ${ch.slug}: unchanged — reusing cached live-chat archive`);
+ return {
+ archive: { slug: ch.slug, archivePath, liveChatCount: prev.count },
+ reused: true,
+ };
+ }
+ }
+ const staging = path.join(stagingDir(opts.paths), ch.slug);
try {
const { linkedCount } = await stageChannel(
opts.paths,
@@ -219,9 +263,10 @@ export async function archiveLiveChat(
);
if (linkedCount === 0) {
await rm(staging, { recursive: true, force: true });
- skipped.push({ slug: ch.slug, reason: "no normalized live chat" });
+ await rm(archivePath, { force: true }).catch(() => {});
+ await rm(archiveSidecarPath(archivePath), { force: true }).catch(() => {});
log(` ${ch.slug}: skipped (no live chat to archive)`);
- continue;
+ return { skipped: { slug: ch.slug, reason: "no normalized live chat" } };
}
try {
await compressArchive(
@@ -233,17 +278,37 @@ export async function archiveLiveChat(
} finally {
await rm(staging, { recursive: true, force: true });
}
+ if (sig) await writeArchiveSidecar(archivePath, sig, linkedCount);
log(
` ${ch.slug}: ${linkedCount} live chats → ${path.basename(archivePath)}`,
);
- archives.push({ slug: ch.slug, archivePath, liveChatCount: linkedCount });
- totalLiveChats += linkedCount;
+ return {
+ archive: { slug: ch.slug, archivePath, liveChatCount: linkedCount },
+ };
} catch (err) {
await rm(staging, { recursive: true, force: true }).catch(() => {});
- skipped.push({ slug: ch.slug, reason: (err as Error).message });
log(` ! ${ch.slug}: ${(err as Error).message}`);
+ return { skipped: { slug: ch.slug, reason: (err as Error).message } };
+ }
+ };
+
+ try {
+ const results = await Promise.all(
+ target.map((ch) => channelLimit(() => buildOne(ch))),
+ );
+ for (const r of results) {
+ if (r.archive) {
+ archives.push(r.archive);
+ totalLiveChats += r.archive.liveChatCount;
+ if (r.reused) reused++;
+ } else if (r.skipped) {
+ skipped.push(r.skipped);
+ }
}
+ } finally {
+ if (!opts.signer) await signer.close();
}
+ if (reused > 0) log(`Reused ${reused} unchanged cached live-chat archive(s).`);
log("");
log(
diff --git a/common/controller/archiveTranscripts.ts b/common/controller/archiveTranscripts.ts
@@ -21,6 +21,7 @@
import path from "node:path";
import {
+ access,
link,
mkdir,
readdir,
@@ -32,6 +33,7 @@ import {
import { execa } from "execa";
import pLimit from "p-limit";
import { listChannels, type ChannelStat } from "./channels";
+import { openChannelSigner, type ChannelSigner } from "../lib/channelSignature";
import {
normalizeTranscript,
type NormalizedTranscript,
@@ -58,12 +60,21 @@ export type ArchiveTranscriptsOptions = {
onLog?: (msg: string) => void;
signal?: AbortSignal;
concurrency?: number;
+ // How many channels' archives to build concurrently. Each channel shells out
+ // to a `zip`/`tar` subprocess, so a few in parallel uses idle cores; kept
+ // modest to avoid thrashing disk on a large corpus. Staging within a channel
+ // stays bounded by `concurrency`.
+ channelConcurrency?: number;
channelSlugs?: string[];
build?: Partial<ArchiveBuildOptions>;
// Destination for the finished archives. Defaults to transcripts/export/
// archives (the standalone editor actions' location); compose-site passes the
// served public/archives dir so the zips ship with the site build.
outDir?: string;
+ // Reuse an already-open per-channel signer (see channelSignature.ts) so the
+ // caller can share one LMDB view across the transcripts + live-chat passes.
+ // When omitted, one is opened for the duration of this call.
+ signer?: ChannelSigner;
};
export type ArchiveTranscriptsResult = {
@@ -86,6 +97,15 @@ function archivesDir(paths: Paths): string {
return path.join(paths.transcriptsDir, "export", "archives");
}
+// The persistent, shared archive cache dir (the default archive outDir). It
+// lives outside any site's public/ tree and survives across builds, so
+// signature-gated archives can be reused build-to-build and across sites.
+// compose-site builds into this cache, then copies each site's member subset
+// into that site's public/archives.
+export function archiveCacheDir(paths: Paths): string {
+ return archivesDir(paths);
+}
+
function stagingDir(paths: Paths): string {
return path.join(paths.transcriptsDir, "export", ".staging");
}
@@ -226,6 +246,79 @@ export function archiveExtension(format: ArchiveFormat): string {
return format;
}
+// --- Incremental cache: skip re-zipping a channel whose content is unchanged --
+// Each produced archive gets a sibling `<archive>.sig` sidecar recording the
+// channel content signature it was built from (plus the entry count, which the
+// Downloads manifest needs and which we'd otherwise have to recompute on a
+// cache hit). A build reuses the existing zip when the current signature matches
+// and the zip is still present.
+
+export function archiveSidecarPath(archivePath: string): string {
+ return `${archivePath}.sig`;
+}
+
+export async function readArchiveSidecar(
+ archivePath: string,
+): Promise<{ signature: string; count: number } | null> {
+ try {
+ const parsed = JSON.parse(
+ await readFile(archiveSidecarPath(archivePath), "utf8"),
+ );
+ if (
+ parsed &&
+ typeof parsed.signature === "string" &&
+ typeof parsed.count === "number"
+ ) {
+ return { signature: parsed.signature, count: parsed.count };
+ }
+ return null;
+ } catch {
+ return null;
+ }
+}
+
+export async function writeArchiveSidecar(
+ archivePath: string,
+ signature: string,
+ count: number,
+): Promise<void> {
+ await writeFile(
+ archiveSidecarPath(archivePath),
+ JSON.stringify({ signature, count }) + "\n",
+ );
+}
+
+async function fileExists(p: string): Promise<boolean> {
+ try {
+ await access(p);
+ return true;
+ } catch {
+ return false;
+ }
+}
+
+// The non-mtime inputs that also determine a channel's archive bytes: the build
+// options and the config fields baked into channel.json. Folded into the
+// signature so a settings/config change forces a rebuild even when no video
+// mtime moved.
+export function archiveSignatureExtra(
+ build: ArchiveBuildOptions,
+ ch: ChannelStat,
+): string {
+ return JSON.stringify({
+ format: build.format,
+ level: build.compressionLevel,
+ meta: build.includeMetadata,
+ pretty: build.prettyPrint,
+ cfg: {
+ name: ch.config.name ?? null,
+ handling: ch.config.handling,
+ platform: ch.config.platform ?? null,
+ url: ch.config.url ?? null,
+ },
+ });
+}
+
// Compress a single staged top-level entry (a slug dir for per-channel
// mode, or one of multiple entries for combined mode) into archivePath.
export async function compressArchive(
@@ -294,16 +387,36 @@ export async function archiveTranscripts(
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
- for (const ch of target) {
- if (opts.signal?.aborted) {
- log("Aborted.");
- break;
- }
- const staging = path.join(stagingDir(opts.paths), ch.slug);
+ const signer = opts.signer ?? openChannelSigner(opts.paths);
+ const channelLimit = pLimit(opts.channelConcurrency ?? 4);
+ let reused = 0;
+
+ // Build (or reuse) one channel's archive. Returns exactly one of archive /
+ // skipped so the caller can reduce results deterministically.
+ const buildOne = async (
+ ch: ChannelStat,
+ ): Promise<{
+ archive?: ArchiveTranscriptsResult["archives"][number];
+ skipped?: ArchiveTranscriptsResult["skipped"][number];
+ reused?: boolean;
+ }> => {
+ if (opts.signal?.aborted) return {};
const archivePath = path.join(
outDir,
`${ch.slug}.${archiveExtension(build.format)}`,
);
+ const sig = signer.signature(ch.slug, archiveSignatureExtra(build, ch));
+ if (sig) {
+ const prev = await readArchiveSidecar(archivePath);
+ if (prev && prev.signature === sig && (await fileExists(archivePath))) {
+ log(` ${ch.slug}: unchanged — reusing cached archive`);
+ return {
+ archive: { slug: ch.slug, archivePath, transcriptCount: prev.count },
+ reused: true,
+ };
+ }
+ }
+ const staging = path.join(stagingDir(opts.paths), ch.slug);
try {
const { linkedCount } = await stageChannel(
opts.paths,
@@ -315,9 +428,12 @@ export async function archiveTranscripts(
);
if (linkedCount === 0) {
await rm(staging, { recursive: true, force: true });
- skipped.push({ slug: ch.slug, reason: "no normalized transcripts" });
+ // A channel that lost all transcripts must stop being served: drop any
+ // stale cached zip + sidecar so it isn't reused next build.
+ await rm(archivePath, { force: true }).catch(() => {});
+ await rm(archiveSidecarPath(archivePath), { force: true }).catch(() => {});
log(` ${ch.slug}: skipped (no transcripts to archive)`);
- continue;
+ return { skipped: { slug: ch.slug, reason: "no normalized transcripts" } };
}
try {
await compressArchive(
@@ -329,17 +445,37 @@ export async function archiveTranscripts(
} finally {
await rm(staging, { recursive: true, force: true });
}
+ if (sig) await writeArchiveSidecar(archivePath, sig, linkedCount);
log(
` ${ch.slug}: ${linkedCount} transcripts → ${path.basename(archivePath)}`,
);
- archives.push({ slug: ch.slug, archivePath, transcriptCount: linkedCount });
- totalTranscripts += linkedCount;
+ return {
+ archive: { slug: ch.slug, archivePath, transcriptCount: linkedCount },
+ };
} catch (err) {
await rm(staging, { recursive: true, force: true }).catch(() => {});
- skipped.push({ slug: ch.slug, reason: (err as Error).message });
log(` ! ${ch.slug}: ${(err as Error).message}`);
+ return { skipped: { slug: ch.slug, reason: (err as Error).message } };
+ }
+ };
+
+ try {
+ const results = await Promise.all(
+ target.map((ch) => channelLimit(() => buildOne(ch))),
+ );
+ for (const r of results) {
+ if (r.archive) {
+ archives.push(r.archive);
+ totalTranscripts += r.archive.transcriptCount;
+ if (r.reused) reused++;
+ } else if (r.skipped) {
+ skipped.push(r.skipped);
+ }
}
+ } finally {
+ if (!opts.signer) await signer.close();
}
+ if (reused > 0) log(`Reused ${reused} unchanged cached archive(s).`);
log("");
log(
diff --git a/common/lib/channelSignature.ts b/common/lib/channelSignature.ts
@@ -0,0 +1,89 @@
+// A per-channel content signature for incremental export builds.
+//
+// The archive + compose stages redo work per channel every build. To skip a
+// channel whose content hasn't changed we need a stable answer to "did channel
+// X change since we last built its output?". There is no per-channel fingerprint
+// in the index, but build-index DOES persist per-VIDEO freshness in the `mtimes`
+// LMDB sub-DB (key [channelSlug, videoDir] -> {metaMs, transcriptMs, subsMs,…}).
+// Folding a channel's mtime records into one hash reuses the exact freshness
+// truth the rest of the build already trusts — so "unchanged" here means the
+// same thing build-index means, and it sidesteps the mtime-drift-across-shards
+// hazard that a raw fs.stat would reintroduce (the LMDB mtimes are the
+// reconciled source, captured once at index time).
+//
+// The signature is over INPUTS, never the produced bytes: archive zips embed a
+// `generatedAt` timestamp (archiveTranscripts.ts channelManifest) so they are
+// not byte-reproducible; only an input signature is stable across builds.
+
+import { createHash } from "node:crypto";
+import { existsSync } from "node:fs";
+import { open } from "lmdb";
+import type { Paths } from "./paths";
+
+type MtimeRecord = {
+ metaMs: number;
+ transcriptMs: number | null;
+ subsMs: number | null;
+};
+type PathKey = [string, string];
+
+export type ChannelSigner = {
+ // A hex signature for the channel's on-disk content, or null when it can't be
+ // determined (no index yet) — callers should treat null as "cannot cache,
+ // rebuild unconditionally". `extra` folds in caller-specific inputs that also
+ // affect the output (e.g. archive build options, channel config).
+ signature(slug: string, extra?: string): string | null;
+ close(): Promise<void>;
+};
+
+// Open a read-only view of the index's per-video mtimes for signature lookups.
+// Returns a no-op signer (signature() -> null) when the index doesn't exist yet
+// or can't be opened, so a fresh checkout simply builds everything.
+export function openChannelSigner(paths: Paths): ChannelSigner {
+ if (!existsSync(paths.lmdbPath)) {
+ return { signature: () => null, close: async () => {} };
+ }
+ let root: ReturnType<typeof open>;
+ try {
+ root = open({ path: paths.lmdbPath, readOnly: true, maxDbs: 14 });
+ } catch {
+ return { signature: () => null, close: async () => {} };
+ }
+ const mtimes = root.openDB<MtimeRecord, PathKey>({
+ name: "mtimes",
+ encoding: "msgpack",
+ });
+ const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ // Fold the index schema version in so a schema bump (which rewrites the whole
+ // index) invalidates every cached signature. Deliberately NOT the `generation`
+ // counter — that bumps whenever ANY channel changes, which would needlessly
+ // invalidate every OTHER channel's signature.
+ const schema = String((meta.get("schema") as number | undefined) ?? "none");
+
+ return {
+ signature(slug: string, extra = ""): string | null {
+ const h = createHash("sha1");
+ h.update(`schema:${schema}\n`);
+ if (extra) h.update(`extra:${extra}\n`);
+ let sawAny = false;
+ for (const { key, value } of mtimes.getRange({
+ start: [slug],
+ end: [slug, ""],
+ })) {
+ const k = key as PathKey;
+ if (k[0] !== slug) break;
+ sawAny = true;
+ h.update(
+ `${k[1]}\t${value.metaMs}\t${value.transcriptMs ?? ""}\t${value.subsMs ?? ""}\n`,
+ );
+ }
+ // A channel with zero indexed videos still gets a stable signature (schema
+ // + extra), so an empty channel can be cached like any other.
+ void sawAny;
+ return h.digest("hex");
+ },
+ async close() {
+ await root.close();
+ },
+ };
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Deploys no longer re-upload unchanged oversize archives to R2.** The deploy step used to stream every over-cap archive to R2 on every deploy, even ones byte-identical to what was already there. Because the export build now reuses an unchanged channel's cached zip verbatim, the upload step first does a cheap `HeadObject` and skips any archive whose R2 object already has the same size — so a redeploy after changing one channel only re-uploads that channel's oversize bundle. See `editor/app/deploy/buildDeployCore.ts`.
- **New "Search aliases" page for authoring known-term suggestions.** A concept is often spelled many ways — transcription in particular mangles them (AI expands `loli` to `lolly`/`loly`) — so searching one spelling silently misses the rest. This page authors a curated dictionary where one entry groups all the trigger spellings of a concept with a single robust replacement (e.g. `\blol(i|ly)`). A **Global** section applies to every site and ships a small seeded default set you can edit or remove; a **per-site** section (shown when a site is selected in the sidebar) adds to or overrides the global list by id — reuse a global id and mark it disabled to hide that alias for just that site. Nothing is forced: the entries only surface a *suggestion* in the viewer's search box, which the searcher can apply or ignore. Stored as `search-aliases.json` (global under the data dir, per-site under `sites/<id>/`) and merged into each site's bundle at build time. See `editor/app/aliases/{page.tsx,actions.ts,EditorAliasesClient.tsx}`, `common/lib/{searchAliases,aliasesStore}.ts`, `editor/app/lib/nav.ts`, and `editor/e2e/aliases.spec.ts`.
- **Fixed: the editor (and every public site) always loaded in light mode until you toggled the theme.** Dark styling is driven purely by a `.dark` class on `<html>`; the pre-paint `<ThemeScript>` set it correctly before first paint, but `<html>` is server-rendered with a static class that omits `dark`, so that class was lost across React's hydration boundary and nothing put it back — the page fell to the light palette on every load until a manual toggle re-applied it directly (which is why toggling then "stuck"). `ThemeProvider` now re-asserts the persisted, resolved family + mode to `<html>` on mount via `useLayoutEffect` (before paint) — idempotent with the script, so there's no flash and no toggle needed. Explicit dark, `system` on a dark OS, and non-Base families all now survive a refresh. See `common/components/ThemeProvider.tsx` and `editor/e2e/theme.spec.ts`.
- **Search results scroll smoothly again on large result sets.** The results list windows one card per matching video (only the on-screen cards are mounted), but each visible card and every one of its hit rows was re-rendering on *every* scroll frame — and each hit row re-ran its `<mark>` highlighting, so a single video with hundreds of hits meant hundreds of redundant highlight passes per frame while scrolling. The result cards and individual hit rows are now memoized so an unchanged card/row is skipped during scroll, and opening the modal on a hit only re-renders the two rows whose highlight state actually changes. No visible/behavioral change — same DOM, same results, just far less work per frame. See `common/components/TranscriptSearch.tsx` (`ResultCard`/`HitRow` memoization, `openWithMode` stabilized via `useCallback`).
diff --git a/editor/app/deploy/buildDeployCore.ts b/editor/app/deploy/buildDeployCore.ts
@@ -8,7 +8,7 @@ import path from "node:path";
import { readdir } from "node:fs/promises";
import { createReadStream } from "node:fs";
import { stat } from "node:fs/promises";
-import { S3Client } from "@aws-sdk/client-s3";
+import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
import { Upload } from "@aws-sdk/lib-storage";
import { runChildIntoLog } from "yt-dlp-transcript-common/jobs/runChild";
import type { Paths } from "yt-dlp-transcript-common/lib/paths";
@@ -157,12 +157,32 @@ export async function runArchiveUploadIntoLog(
`[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`,
);
try {
+ let uploaded = 0;
+ let skipped = 0;
for (const file of files) {
if (signal.aborted) return 1;
const key = `${site.siteId}/archives/${file}`;
const filePath = path.join(stagingDir, file);
const { size } = await stat(filePath);
+ // Skip the re-upload when R2 already holds an object of the same size for
+ // this key. The archive cache makes an unchanged channel's zip byte-stable
+ // build-to-build, so a same-size object is the same object; a changed
+ // channel re-zips to a different size. Avoids re-streaming unchanged
+ // multi-MB archives on every deploy. HeadObject is a cheap metadata call.
+ try {
+ const head = await client.send(
+ new HeadObjectCommand({ Bucket: bucket, Key: key }),
+ );
+ if (head.ContentLength === size) {
+ skipped++;
+ onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`);
+ continue;
+ }
+ } catch {
+ // Not found (or HEAD not permitted) → fall through and upload.
+ }
onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`);
+ uploaded++;
const upload = new Upload({
client,
params: {
@@ -184,6 +204,10 @@ export async function runArchiveUploadIntoLog(
}
onLog(`[archives] ${key} done.\n`);
}
+ onLog(
+ `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`,
+ );
+ return 0;
} catch (err) {
onLog(
`[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`,
@@ -192,8 +216,6 @@ export async function runArchiveUploadIntoLog(
} finally {
client.destroy();
}
- onLog("[archives] R2 upload complete.\n");
- return 0;
}
// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,8 @@
# Changelog
+## [Unreleased]
+- **Rebuilds now skip channels that haven't changed.** The export build used to redo almost everything from scratch every run — it re-zipped every channel's transcript and live-chat download bundle, and `rm -rf`'d and re-copied every channel's transcript/subs pages into the served tree — even when nothing about that channel had changed. Each of these is now gated on a cheap per-channel signature: archive zips are built once into a persistent shared cache (`transcripts/export/archives`) keyed by a content signature and reused build-to-build and across sites, and the per-channel page trees are reconciled in place (only changed channels are re-copied, removed channels pruned). A no-change recompose drops from re-zipping/re-copying gigabytes to a few seconds. Output is identical — the signature is over inputs (video mtimes recorded by `build:index`, channel config, and archive options), so a channel is only rebuilt when its content actually changes; a schema bump or settings change invalidates the cache. Also parallelizes the per-channel archive compressor across a few channels at once. See `common/lib/channelSignature.ts`, `common/controller/{archiveTranscripts,archiveLiveChat}.ts`, and `common/bin/compose-site.ts`.
+
## [0.7.2] - 2026-07-07
- **The Downloads page was redesigned around channels.** A channel's transcript bundle and its live-chat bundle now sit together in one card — a catalog "record" headed by the channel name and its total video count and size — instead of scattered, separately-bordered rows. Each bundle shows its item count and filename, and the size moved into a padded download button (fixing the badge that used to butt against the card edge). The nested/mismatched borders are gone, the whole page adapts to every theme, and an oversize bundle still reads as unavailable with its reason.