commit 10b4f62c5918e7a114fc0df16a58483da3a54e47
parent 05877ffb6d9756dbbb8173225125dd05bf403794
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 00:45:27 -0400
common: buildDeployCore moves to the core as publish/build.ts
`editor/app/sites/lib/buildDeployCore.ts` imported nothing from the editor,
so it moves unchanged to `common/publish/build.ts` (one-core Phase 4 item 1).
The two callers, `sites/lib/buildAction.ts` and `deployAction.ts`, import
`yt-dlp-transcript-common/publish/build`; no re-export stays behind, since no
spec imports the old path. Log lines, exit codes and paths are identical.
`@aws-sdk/client-s3` and `@aws-sdk/lib-storage` move from editor's
dependencies to common's; the lockfile change is those two importer entries
and nothing else (`pnpm install --frozen-lockfile` passes).
common/package.json gains `./publish/*` in `exports` (the editor resolves
through it) and `publish` in the test glob. architecture.test.ts learns the
layer: `publish/` may not import `views/` or `components/`, and `lib/` and
`components/` may not import `publish/`. A unit test pins the three pure
path helpers.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
10 files changed, 648 insertions(+), 594 deletions(-)
diff --git a/DEPLOY_CLOUDFLARE.md b/DEPLOY_CLOUDFLARE.md
@@ -242,7 +242,7 @@ small and commented if you want to adjust them.
The archive `Cache-Control` is also set at upload time (`Cache-Control: public,
max-age=3600`, constant `ARCHIVE_CACHE_CONTROL` in
-`editor/app/deploy/buildDeployCore.ts`); the Worker reads it back when caching. Archive
+`common/publish/build.ts`); the Worker reads it back when caching. Archive
filenames are stable and overwritten in place on re-deploy, so this 1-hour bound is what
keeps a re-uploaded archive from being served stale for long — raise it if your archives
rarely change.
diff --git a/common/architecture.test.ts b/common/architecture.test.ts
@@ -35,15 +35,20 @@ const FORBIDDEN: Record<string, readonly string[]> = {
// searchQuery for a mode union. All four inverted — the pipeline and the two
// types moved down, and the fetch and the memo are now injected — so the
// guard turns on with nothing added to the allow-list to pay for it.
- lib: ["controller", "jobs", "components", "views"],
+ lib: ["controller", "jobs", "components", "views", "publish"],
jobs: ["controller", "views"],
controller: ["views"],
- components: ["controller", "jobs", "ytdlp"],
+ components: ["controller", "jobs", "ytdlp", "publish"],
// `views/` is the view-model layer (one-core phase 3 slice 1): pure functions
// that fold live state into a payload. It sits ABOVE dispatch and BELOW ui,
// so it may read lib/, controller/ and jobs/ for pure helpers and types, and
// may not reach up into components/ or sideways into the entry points.
views: ["components", "ytdlp", "bin", "social"],
+ // `publish/` is the publish layer (one-core phase 4 slice 1): building a site,
+ // uploading its archives, deploying it. It sits above dispatch and below views,
+ // so it may read lib/, jobs/ and controller/, and may not reach up into
+ // views/ or components/.
+ publish: ["views", "components"],
};
// The directories walked. `bin/` and `social/` are scanned so a back-edge cannot
@@ -57,6 +62,7 @@ const ROOTS = [
"ytdlp",
"bin",
"views",
+ "publish",
];
// Today's back-edges, `<file> -> <imported module>`, each with why it is still
@@ -164,10 +170,10 @@ test("no new back-edges between common's layers", async () => {
assert.deepEqual(
unexpected,
[],
- `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, ` +
+ `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, publish/, ` +
`components/ or views/; ` +
`jobs/ and controller/ may not import views/; components/ may not ` +
- `import controller/, jobs/ or ytdlp/; views/ may not import ` +
+ `import controller/, jobs/, ytdlp/ or publish/; publish/ may not import views/ or components/; views/ may not import ` +
`components/. Move the type or the function down a ` +
`layer instead of adding it to ALLOWED. Found: ${unexpected.join(", ")}`,
);
diff --git a/common/package.json b/common/package.json
@@ -35,6 +35,7 @@
"./controller/*": "./controller/*.ts",
"./jobs/*": "./jobs/*.ts",
"./views/*": "./views/*.ts",
+ "./publish/*": "./publish/*.ts",
"./social/*": "./social/*.ts",
"./ytdlp/*": "./ytdlp/*.ts",
"./bin/*": "./bin/*.ts",
@@ -42,9 +43,11 @@
"./styles/*": "./styles/*.ts"
},
"scripts": {
- "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*/*.test.ts\""
+ "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*/*.test.ts\""
},
"dependencies": {
+ "@aws-sdk/client-s3": "^3.1080.0",
+ "@aws-sdk/lib-storage": "^3.1080.0",
"@sindresorhus/slugify": "^3.0.0",
"@tanstack/react-query": "^5.99.1",
"@tanstack/react-virtual": "^3.13.0",
diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts
@@ -0,0 +1,43 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Paths } from "../lib/paths";
+import {
+ dockerSiteOutDir,
+ dockerSiteStagingDir,
+ resolveOutDir,
+} from "./build";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// The three pure path helpers of the publish layer. Their outputs are the
+// on-disk contract the build and deploy phases share (and that
+// `docker/build-site.sh` mounts), so they are pinned here exactly.
+
+const paths = {
+ exportDir: "/repo/export",
+ exportBuildsDir: "/repo/export/.export-builds",
+} as Paths;
+
+test("resolveOutDir is export/out whatever the site", () => {
+ assert.equal(resolveOutDir("jeralyzer", paths), "/repo/export/out");
+ assert.equal(resolveOutDir("anilyzer", paths), "/repo/export/out");
+});
+
+test("dockerSiteOutDir is a per-site out/ under exportBuildsDir", () => {
+ assert.equal(
+ dockerSiteOutDir(paths, "jeralyzer"),
+ "/repo/export/.export-builds/jeralyzer/out",
+ );
+ assert.notEqual(
+ dockerSiteOutDir(paths, "jeralyzer"),
+ dockerSiteOutDir(paths, "anilyzer"),
+ );
+});
+
+test("dockerSiteStagingDir nests .r2-staging/<site>/archives under the site's build dir", () => {
+ assert.equal(
+ dockerSiteStagingDir(paths, "jeralyzer"),
+ "/repo/export/.export-builds/jeralyzer/.r2-staging/jeralyzer/archives",
+ );
+});
diff --git a/common/publish/build.ts b/common/publish/build.ts
@@ -0,0 +1,582 @@
+// The publish layer's build/deploy primitives: one site's host build, the
+// docker per-site fan-out, the R2 archive upload and the Pages deploy. Moved
+// here from the editor (`editor/app/sites/lib/buildDeployCore.ts`) in one-core
+// Phase 4 slice 1, unchanged; the editor's build/deploy server actions
+// (`editor/app/sites/lib/{buildAction,deployAction}.ts`) call them as jobs.
+//
+// These take an `onLog` callback and an AbortSignal, so they CANNOT live in a
+// "use server" module (every export there becomes a server action, which
+// forbids non-serializable args). Keep them as plain helpers.
+
+import path from "node:path";
+import { mkdir, readdir, stat } from "node:fs/promises";
+import { createReadStream, existsSync } from "node:fs";
+import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
+import { Upload } from "@aws-sdk/lib-storage";
+import { runChildIntoLog } from "../jobs/runChild";
+import {
+ deploymentUrlIn,
+ pagesDeployArgs,
+ previewAliasUrl,
+} from "../lib/pagesDeploy";
+import type { Paths } from "../lib/paths";
+import { getSettings } from "../lib/settings";
+import type { Site } from "../lib/site";
+
+// Where the basic (host) build writes the static bundle to deploy: the fixed
+// export/out, composed one site at a time. The docker fan-out writes per-site
+// out/ dirs instead — see dockerSiteOutDir.
+export function resolveOutDir(_siteId: string, paths: Paths): string {
+ return path.join(paths.exportDir, "out");
+}
+
+// Where a docker per-site container writes its built bundle (mounted as /site/out
+// inside the container). Isolated per site so parallel builds never collide.
+export function dockerSiteOutDir(paths: Paths, siteId: string): string {
+ return path.join(paths.exportBuildsDir, siteId, "out");
+}
+
+// Where a docker per-site container stages oversize archives for R2 upload. The
+// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/
+// public), so compose writes .r2-staging as its sibling under exportBuildsDir.
+export function dockerSiteStagingDir(paths: Paths, siteId: string): string {
+ return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives");
+}
+
+// Run the basic (host) build phase for one site, streaming into `onLog`,
+// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream
+// on the build queue, since the export/ tree is shared). This is the single-site
+// build for basic mode, and the fallback the all-sites docker action drops to
+// when no container engine is available. The parallel per-site container path
+// lives in runDockerBuildAllPhase.
+export async function runBuildPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ siteId: string,
+ paths: Paths,
+ opts?: { skipData?: boolean; skipArchives?: boolean },
+): Promise<number> {
+ // When skipping the data rebuild, run the `build:nodata` script instead of
+ // `build`. `build:nodata` has the same body (compose:site + next build) but a
+ // different name, so npm's `prebuild` lifecycle hook (which runs build:data =
+ // index/stats/templates) does NOT fire — we compose from the existing
+ // .export-index staging. Assumes a prior full build produced that staging.
+ const skipData = opts?.skipData === true;
+ if (skipData) {
+ onLog(
+ "[notice] Skipping data rebuild (index/stats/charts) — composing from " +
+ "existing .export-index staging.\n",
+ );
+ }
+ // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes
+ // compose-site skip generation this build regardless of the global/site flags.
+ const skipArchives = opts?.skipArchives === true;
+ if (skipArchives) {
+ onLog("[notice] Skipping archive-zip generation for this build.\n");
+ }
+ return runChildIntoLog(onLog, signal, {
+ command: "pnpm",
+ args: ["run", skipData ? "build:nodata" : "build"],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ EXPORT_PUBLIC_DIR: paths.exportPublicDir,
+ SITE_ID: siteId,
+ ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}),
+ },
+ });
+}
+
+// Where the basic (host) compose staged this site's oversize archives for R2
+// upload. Kept outside export/public so they never ship as Pages assets. Mirrors
+// the path composeArchives writes to in common/bin/compose-site.ts. The docker
+// fan-out uses dockerSiteStagingDir instead, passed explicitly.
+function archiveStagingDir(siteId: string, paths: Paths): string {
+ return path.join(
+ path.dirname(paths.exportPublicDir),
+ ".r2-staging",
+ siteId,
+ "archives",
+ );
+}
+
+// Said once at the top of every preview deploy, because the one thing a preview
+// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a
+// preview's oversize archives overwrite the keys production's manifest points
+// at. That is cheap and harmless in practice — the upload skips any object R2
+// already holds at the same size, and an unchanged channel re-zips byte-stable
+// — but "harmless because of a size check" is exactly the kind of thing an
+// operator should be told rather than left to discover.
+export const PREVIEW_SHARES_ARCHIVES_NOTICE =
+ "[notice] A preview shares the production R2 archive bucket — unchanged " +
+ "archives are skipped, so this is cheap, but a CHANGED archive replaces the " +
+ "one production links to.\n";
+
+// Cache-Control set on every uploaded archive. Served through a Cloudflare custom
+// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead
+// of hitting R2 (each origin GET is a billable Class B op), which is the main cost
+// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood
+// absorption against re-deployed archives (stable filenames, overwritten in place)
+// going stale; raise it if your archives rarely change.
+const ARCHIVE_CACHE_CONTROL = "public, max-age=3600";
+
+// MIME type stored on each uploaded archive. Archives are `.zip` bundles.
+const ARCHIVE_CONTENT_TYPE = "application/zip";
+
+// Upload this site's staged oversize archives to the configured R2 bucket, so the
+// remote URLs the served manifest points at actually resolve. Uploads go through
+// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) —
+// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real
+// live-chat archives blow past, whereas multipart streams any size. Keys match
+// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must
+// run BEFORE the Pages deploy so the manifest never points at a missing object.
+//
+// No-op (returns 0) when overflow storage isn't configured or nothing was staged.
+// Credentials come from the host env (NOT the settings JSON — secrets don't
+// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID
+// (for the S3 endpoint). If a bucket is configured and files are staged but the
+// credentials are missing, we fail (return non-zero) so the deploy aborts rather
+// than shipping a manifest that points at un-uploaded objects.
+export async function runArchiveUploadIntoLog(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ site: Site,
+ paths: Paths,
+ // Where the oversize archives were staged. Defaults to the basic host location;
+ // the docker deploy phase passes dockerSiteStagingDir since its container wrote
+ // .r2-staging under exportBuildsDir/<siteId> instead.
+ stagingDirOverride?: string,
+): Promise<number> {
+ const bucket = getSettings().archiveStorage?.bucket?.trim();
+ if (!bucket) return 0;
+
+ const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths);
+ let files: string[];
+ try {
+ files = (await readdir(stagingDir)).filter((f) => !f.startsWith("."));
+ } catch {
+ // No staging dir → nothing oversize this build.
+ return 0;
+ }
+ if (files.length === 0) return 0;
+
+ const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim();
+ const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim();
+ const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim();
+ if (!accessKeyId || !secretAccessKey || !accountId) {
+ onLog(
+ `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` +
+ `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` +
+ `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` +
+ `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` +
+ `missing files.\n`,
+ );
+ return 1;
+ }
+
+ const client = new S3Client({
+ region: "auto",
+ endpoint: `https://${accountId}.r2.cloudflarestorage.com`,
+ credentials: { accessKeyId, secretAccessKey },
+ });
+
+ onLog(
+ `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`,
+ );
+ try {
+ let uploaded = 0;
+ let skipped = 0;
+ for (const file of files) {
+ if (signal.aborted) return 1;
+ const key = `${site.siteId}/archives/${file}`;
+ const filePath = path.join(stagingDir, file);
+ const { size } = await stat(filePath);
+ // Skip the re-upload when R2 already holds an object of the same size for
+ // this key. The archive cache makes an unchanged channel's zip byte-stable
+ // build-to-build, so a same-size object is the same object; a changed
+ // channel re-zips to a different size. Avoids re-streaming unchanged
+ // multi-MB archives on every deploy. HeadObject is a cheap metadata call.
+ try {
+ const head = await client.send(
+ new HeadObjectCommand({ Bucket: bucket, Key: key }),
+ );
+ if (head.ContentLength === size) {
+ skipped++;
+ onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`);
+ continue;
+ }
+ } catch {
+ // Not found (or HEAD not permitted) → fall through and upload.
+ }
+ onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`);
+ uploaded++;
+ const upload = new Upload({
+ client,
+ params: {
+ Bucket: bucket,
+ Key: key,
+ Body: createReadStream(filePath),
+ ContentType: ARCHIVE_CONTENT_TYPE,
+ CacheControl: ARCHIVE_CACHE_CONTROL,
+ },
+ });
+ const onAbort = () => {
+ upload.abort().catch(() => {});
+ };
+ signal.addEventListener("abort", onAbort, { once: true });
+ try {
+ await upload.done();
+ } finally {
+ signal.removeEventListener("abort", onAbort);
+ }
+ onLog(`[archives] ${key} done.\n`);
+ }
+ onLog(
+ `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`,
+ );
+ return 0;
+ } catch (err) {
+ onLog(
+ `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`,
+ );
+ return 1;
+ } finally {
+ client.destroy();
+ }
+}
+
+// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare
+// Pages project, streaming into `onLog`, returning the exit code. Runs on the
+// host with the host's Cloudflare credentials (process.env) — deploy never runs
+// inside a container, so container wrangler auth is never needed.
+//
+// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to
+// any branch but the project's production branch as a preview, reachable at the
+// branch alias. Omitting it leaves the argv byte-identical to what production
+// has always run — no `--branch`, so wrangler infers the branch from the
+// checkout, which is the long-standing behaviour (and the long-standing hazard:
+// a "production" deploy run from a non-main checkout silently becomes a
+// preview).
+//
+// Either way, the URL wrangler prints ("Take a peek over at …") earns one
+// terminal line of its own, because the streamed log scrolls and an operator
+// who looked away has nowhere else to find it. A preview also gets the stable
+// branch alias, which is knowable without reading the log at all.
+export async function runDeployIntoLog(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ site: Site,
+ outDir: string,
+ paths: Paths,
+ opts?: { previewBranch?: string },
+): Promise<number> {
+ const project = site.cloudflareProject as string;
+ const previewBranch = opts?.previewBranch?.trim() || undefined;
+
+ // Spot the deployment URL as it streams past rather than re-reading the
+ // finished log file: the log is the operator's too, and buffering it a second
+ // time to grep it would double a big deploy's memory for one line of output.
+ let deploymentUrl: string | null = null;
+ const watch = (line: string) => {
+ if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project);
+ onLog(line);
+ };
+
+ const code = await runChildIntoLog(watch, signal, {
+ command: "pnpm",
+ args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ EXPORT_PUBLIC_DIR: paths.exportPublicDir,
+ SITE_ID: site.siteId,
+ },
+ });
+
+ // Only on success. A URL scraped out of a failed run points at nothing — or
+ // worse, at the deployment that is still live.
+ if (code === 0) {
+ if (previewBranch) {
+ const alias = previewAliasUrl(project, previewBranch);
+ onLog(
+ `[preview] ${alias}` +
+ (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") +
+ "\n",
+ );
+ } else if (deploymentUrl) {
+ onLog(`[deployed] ${deploymentUrl}\n`);
+ }
+ }
+ return code;
+}
+
+// ---------------------------------------------------------------------------
+// Docker export pipeline (buildPipeline.mode = "docker")
+//
+// Three ordered phases (see DEPLOY_DOCKER.md):
+// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives
+// (warm the shared archive cache). One writer of the shared state.
+// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next
+// build in its own container, read-only over the shared caches, writing only
+// its per-site out/ under exportBuildsDir/<siteId>.
+// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase).
+// ---------------------------------------------------------------------------
+
+// The container engine binary. Defaults to `docker`; podman is a CLI drop-in
+// (and rootless podman yields host-owned outputs without needing `-u`).
+function dockerBin(): string {
+ return process.env.DOCKER_BIN?.trim() || "docker";
+}
+
+export type SiteBuildOutcome = { siteId: string; code: number };
+export type SiteDeployOutcome = {
+ siteId: string;
+ status: "deployed" | "skipped" | "failed";
+ reason?: string;
+};
+
+// Cheap probe: is the container engine installed and its daemon reachable? Used
+// to fall back to serial host builds when docker isn't available.
+export async function dockerAvailable(signal: AbortSignal): Promise<boolean> {
+ const code = await runChildIntoLog(() => {}, signal, {
+ command: dockerBin(),
+ args: ["version"],
+ cwd: process.cwd(),
+ });
+ return code === 0;
+}
+
+// Run a host-side export pnpm script (build:data / build:archives) for Phase A.
+// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the
+// shared index/staging land at their canonical export/.export-index location
+// (exactly what the fan-out containers mount read-only).
+async function runHostScript(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ paths: Paths,
+ script: string,
+ extraEnv?: Record<string, string>,
+): Promise<number> {
+ return runChildIntoLog(onLog, signal, {
+ command: "pnpm",
+ args: ["run", script],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ ...extraEnv,
+ },
+ });
+}
+
+// Build (or reuse cached layers of) the per-site build image.
+async function ensureBuildImage(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ paths: Paths,
+): Promise<number> {
+ const { dockerImage, dockerfile } = getSettings().buildPipeline;
+ onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`);
+ return runChildIntoLog(onLog, signal, {
+ command: dockerBin(),
+ args: ["build", "-f", dockerfile, "-t", dockerImage, "."],
+ cwd: paths.monorepoRoot,
+ });
+}
+
+// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index,
+// and the .export-index staging read-only; mounts the per-site output dir rw.
+// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses
+// next/font/google, which fetches the site's fonts from Google at build time;
+// `--network=none` would fail the build. Isolation still comes from the per-site
+// output dir, the read-only shared mounts, and the non-root `-u` user.
+async function runDockerBuildOne(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ siteId: string,
+ paths: Paths,
+ opts?: { skipArchives?: boolean },
+): Promise<number> {
+ const { dockerImage } = getSettings().buildPipeline;
+ const siteDir = path.join(paths.exportBuildsDir, siteId);
+ // Pre-create the mount target as the host user so container-written files are
+ // host-owned (paired with `-u` below), not created root-owned by the daemon.
+ await mkdir(siteDir, { recursive: true });
+
+ const args = ["run", "--rm", "--init"];
+ const uid = typeof process.getuid === "function" ? process.getuid() : null;
+ const gid = typeof process.getgid === "function" ? process.getgid() : null;
+ if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`);
+ // Optional resource caps so N parallel builds (each next build can use ~8 GB)
+ // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md.
+ const mem = process.env.DOCKER_BUILD_MEMORY?.trim();
+ const cpus = process.env.DOCKER_BUILD_CPUS?.trim();
+ if (mem) args.push("--memory", mem);
+ if (cpus) args.push("--cpus", cpus);
+ args.push(
+ "-v", `${paths.transcriptsDir}:/data/transcripts:ro`,
+ "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`,
+ "-v", `${siteDir}:/site`,
+ "-e", `SITE_ID=${siteId}`,
+ );
+ // Mount the host settings.json fresh (build config: archive storage, size caps)
+ // rather than relying on a possibly-stale copy — it is NOT baked into the image.
+ if (existsSync(paths.settingsFile)) {
+ args.push(
+ "-v", `${paths.settingsFile}:/data/settings.json:ro`,
+ "-e", "SETTINGS_FILE=/data/settings.json",
+ );
+ }
+ // Mount the entrypoint fresh over the baked copy so a script tweak takes effect
+ // without an image rebuild (the image still bakes it as a fallback).
+ const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh");
+ if (existsSync(entrypoint)) {
+ args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`);
+ }
+ if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0");
+ args.push(dockerImage);
+
+ return runChildIntoLog(onLog, signal, {
+ command: dockerBin(),
+ args,
+ cwd: paths.monorepoRoot,
+ label: `[${siteId}] `,
+ });
+}
+
+// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an
+// infrastructure failure (data phase / archive warm / image build) that aborts
+// the whole run before any site could build; per-site build failures are
+// returned, not thrown, so one bad site never blocks the rest.
+export async function runDockerBuildAllPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ sites: Site[],
+ paths: Paths,
+ opts?: { skipArchives?: boolean },
+): Promise<SiteBuildOutcome[]> {
+ const { maxParallelBuilds } = getSettings().buildPipeline;
+
+ // --- Phase A: shared data + archive cache (host, serial) ---
+ onLog("=== Phase A: shared data + archive cache (host) ===");
+ const dataCode = await runHostScript(onLog, signal, paths, "build:data");
+ if (signal.aborted) return [];
+ if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`);
+ if (!opts?.skipArchives) {
+ const archCode = await runHostScript(onLog, signal, paths, "build:archives");
+ if (signal.aborted) return [];
+ if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`);
+ }
+
+ const imgCode = await ensureBuildImage(onLog, signal, paths);
+ if (signal.aborted) return [];
+ if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`);
+
+ // --- Phase B: per-site fan-out (containers, parallel) ---
+ onLog(
+ `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`,
+ );
+ return runWithConcurrency(sites, maxParallelBuilds, async (site) => {
+ if (signal.aborted) return { siteId: site.siteId, code: 1 };
+ const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts);
+ onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`);
+ return { siteId: site.siteId, code };
+ });
+}
+
+// Phase C: deploy each built site SERIALLY on the host, after the build barrier.
+// Partial-failure tolerant — a site that fails to upload/deploy is recorded and
+// the loop continues. Sites that failed to build, or have no Cloudflare project,
+// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site;
+// basic fallback: export/out).
+export async function runDockerDeployAllPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ sites: Site[],
+ builtOk: Set<string>,
+ paths: Paths,
+ outDirFor: (siteId: string) => string,
+): Promise<SiteDeployOutcome[]> {
+ const outcomes: SiteDeployOutcome[] = [];
+ for (const site of sites) {
+ if (signal.aborted) break;
+ if (!builtOk.has(site.siteId)) {
+ onLog(`[${site.siteId}] deploy skipped — build failed`);
+ outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" });
+ continue;
+ }
+ if (!site.cloudflareProject) {
+ onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "skipped",
+ reason: "no cloudflareProject",
+ });
+ continue;
+ }
+ onLog(`=== Deploy ${site.siteId} ===`);
+ const uploadCode = await runArchiveUploadIntoLog(
+ onLog,
+ signal,
+ site,
+ paths,
+ dockerSiteStagingDir(paths, site.siteId),
+ );
+ if (signal.aborted) break;
+ if (uploadCode !== 0) {
+ onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "failed",
+ reason: `R2 upload exit ${uploadCode}`,
+ });
+ continue;
+ }
+ const deployCode = await runDeployIntoLog(
+ onLog,
+ signal,
+ site,
+ outDirFor(site.siteId),
+ paths,
+ );
+ if (signal.aborted) break;
+ if (deployCode !== 0) {
+ onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "failed",
+ reason: `deploy exit ${deployCode}`,
+ });
+ continue;
+ }
+ onLog(`[${site.siteId}] deployed.`);
+ outcomes.push({ siteId: site.siteId, status: "deployed" });
+ }
+ return outcomes;
+}
+
+// Bounded-concurrency map over a fixed work set, preserving input order in the
+// results. No external dep; a fresh worker pulls the next index until exhausted.
+async function runWithConcurrency<T, R>(
+ items: T[],
+ limit: number,
+ worker: (item: T) => Promise<R>,
+): Promise<R[]> {
+ const results: R[] = new Array(items.length);
+ let next = 0;
+ const width = Math.max(1, Math.min(limit, items.length));
+ const runners = Array.from({ length: width }, async () => {
+ while (true) {
+ const i = next++;
+ if (i >= items.length) break;
+ results[i] = await worker(items[i]);
+ }
+ });
+ await Promise.all(runners);
+ return results;
+}
diff --git a/editor/app/sites/lib/buildAction.ts b/editor/app/sites/lib/buildAction.ts
@@ -33,7 +33,7 @@ import {
runDockerDeployAllPhase,
type SiteBuildOutcome,
type SiteDeployOutcome,
-} from "./buildDeployCore";
+} from "yt-dlp-transcript-common/publish/build";
const DEFAULT_BUILD_QUEUE = "build";
const DEPLOY_QUEUE = "deploy";
diff --git a/editor/app/sites/lib/buildDeployCore.ts b/editor/app/sites/lib/buildDeployCore.ts
@@ -1,578 +0,0 @@
-// Shared, server-only build/deploy primitives used by the build/deploy server
-// actions (buildAction.ts, deployAction.ts). These take an `onLog` callback and
-// an AbortSignal, so they CANNOT live in a "use server" module (every export
-// there becomes a server action, which forbids non-serializable args). Keep them
-// here as plain helpers and import them into the thin action wrappers.
-
-import path from "node:path";
-import { mkdir, readdir, stat } from "node:fs/promises";
-import { createReadStream, existsSync } from "node:fs";
-import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
-import { Upload } from "@aws-sdk/lib-storage";
-import { runChildIntoLog } from "yt-dlp-transcript-common/jobs/runChild";
-import {
- deploymentUrlIn,
- pagesDeployArgs,
- previewAliasUrl,
-} from "yt-dlp-transcript-common/lib/pagesDeploy";
-import type { Paths } from "yt-dlp-transcript-common/lib/paths";
-import { getSettings } from "yt-dlp-transcript-common/lib/settings";
-import type { Site } from "yt-dlp-transcript-common/lib/site";
-
-// Where the basic (host) build writes the static bundle to deploy: the fixed
-// export/out, composed one site at a time. The docker fan-out writes per-site
-// out/ dirs instead — see dockerSiteOutDir.
-export function resolveOutDir(_siteId: string, paths: Paths): string {
- return path.join(paths.exportDir, "out");
-}
-
-// Where a docker per-site container writes its built bundle (mounted as /site/out
-// inside the container). Isolated per site so parallel builds never collide.
-export function dockerSiteOutDir(paths: Paths, siteId: string): string {
- return path.join(paths.exportBuildsDir, siteId, "out");
-}
-
-// Where a docker per-site container stages oversize archives for R2 upload. The
-// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/
-// public), so compose writes .r2-staging as its sibling under exportBuildsDir.
-export function dockerSiteStagingDir(paths: Paths, siteId: string): string {
- return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives");
-}
-
-// Run the basic (host) build phase for one site, streaming into `onLog`,
-// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream
-// on the build queue, since the export/ tree is shared). This is the single-site
-// build for basic mode, and the fallback the all-sites docker action drops to
-// when no container engine is available. The parallel per-site container path
-// lives in runDockerBuildAllPhase.
-export async function runBuildPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- siteId: string,
- paths: Paths,
- opts?: { skipData?: boolean; skipArchives?: boolean },
-): Promise<number> {
- // When skipping the data rebuild, run the `build:nodata` script instead of
- // `build`. `build:nodata` has the same body (compose:site + next build) but a
- // different name, so npm's `prebuild` lifecycle hook (which runs build:data =
- // index/stats/templates) does NOT fire — we compose from the existing
- // .export-index staging. Assumes a prior full build produced that staging.
- const skipData = opts?.skipData === true;
- if (skipData) {
- onLog(
- "[notice] Skipping data rebuild (index/stats/charts) — composing from " +
- "existing .export-index staging.\n",
- );
- }
- // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes
- // compose-site skip generation this build regardless of the global/site flags.
- const skipArchives = opts?.skipArchives === true;
- if (skipArchives) {
- onLog("[notice] Skipping archive-zip generation for this build.\n");
- }
- return runChildIntoLog(onLog, signal, {
- command: "pnpm",
- args: ["run", skipData ? "build:nodata" : "build"],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- EXPORT_PUBLIC_DIR: paths.exportPublicDir,
- SITE_ID: siteId,
- ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}),
- },
- });
-}
-
-// Where the basic (host) compose staged this site's oversize archives for R2
-// upload. Kept outside export/public so they never ship as Pages assets. Mirrors
-// the path composeArchives writes to in common/bin/compose-site.ts. The docker
-// fan-out uses dockerSiteStagingDir instead, passed explicitly.
-function archiveStagingDir(siteId: string, paths: Paths): string {
- return path.join(
- path.dirname(paths.exportPublicDir),
- ".r2-staging",
- siteId,
- "archives",
- );
-}
-
-// Said once at the top of every preview deploy, because the one thing a preview
-// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a
-// preview's oversize archives overwrite the keys production's manifest points
-// at. That is cheap and harmless in practice — the upload skips any object R2
-// already holds at the same size, and an unchanged channel re-zips byte-stable
-// — but "harmless because of a size check" is exactly the kind of thing an
-// operator should be told rather than left to discover.
-export const PREVIEW_SHARES_ARCHIVES_NOTICE =
- "[notice] A preview shares the production R2 archive bucket — unchanged " +
- "archives are skipped, so this is cheap, but a CHANGED archive replaces the " +
- "one production links to.\n";
-
-// Cache-Control set on every uploaded archive. Served through a Cloudflare custom
-// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead
-// of hitting R2 (each origin GET is a billable Class B op), which is the main cost
-// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood
-// absorption against re-deployed archives (stable filenames, overwritten in place)
-// going stale; raise it if your archives rarely change.
-const ARCHIVE_CACHE_CONTROL = "public, max-age=3600";
-
-// MIME type stored on each uploaded archive. Archives are `.zip` bundles.
-const ARCHIVE_CONTENT_TYPE = "application/zip";
-
-// Upload this site's staged oversize archives to the configured R2 bucket, so the
-// remote URLs the served manifest points at actually resolve. Uploads go through
-// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) —
-// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real
-// live-chat archives blow past, whereas multipart streams any size. Keys match
-// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must
-// run BEFORE the Pages deploy so the manifest never points at a missing object.
-//
-// No-op (returns 0) when overflow storage isn't configured or nothing was staged.
-// Credentials come from the host env (NOT the settings JSON — secrets don't
-// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID
-// (for the S3 endpoint). If a bucket is configured and files are staged but the
-// credentials are missing, we fail (return non-zero) so the deploy aborts rather
-// than shipping a manifest that points at un-uploaded objects.
-export async function runArchiveUploadIntoLog(
- onLog: (line: string) => void,
- signal: AbortSignal,
- site: Site,
- paths: Paths,
- // Where the oversize archives were staged. Defaults to the basic host location;
- // the docker deploy phase passes dockerSiteStagingDir since its container wrote
- // .r2-staging under exportBuildsDir/<siteId> instead.
- stagingDirOverride?: string,
-): Promise<number> {
- const bucket = getSettings().archiveStorage?.bucket?.trim();
- if (!bucket) return 0;
-
- const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths);
- let files: string[];
- try {
- files = (await readdir(stagingDir)).filter((f) => !f.startsWith("."));
- } catch {
- // No staging dir → nothing oversize this build.
- return 0;
- }
- if (files.length === 0) return 0;
-
- const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim();
- const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim();
- const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim();
- if (!accessKeyId || !secretAccessKey || !accountId) {
- onLog(
- `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` +
- `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` +
- `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` +
- `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` +
- `missing files.\n`,
- );
- return 1;
- }
-
- const client = new S3Client({
- region: "auto",
- endpoint: `https://${accountId}.r2.cloudflarestorage.com`,
- credentials: { accessKeyId, secretAccessKey },
- });
-
- onLog(
- `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`,
- );
- try {
- let uploaded = 0;
- let skipped = 0;
- for (const file of files) {
- if (signal.aborted) return 1;
- const key = `${site.siteId}/archives/${file}`;
- const filePath = path.join(stagingDir, file);
- const { size } = await stat(filePath);
- // Skip the re-upload when R2 already holds an object of the same size for
- // this key. The archive cache makes an unchanged channel's zip byte-stable
- // build-to-build, so a same-size object is the same object; a changed
- // channel re-zips to a different size. Avoids re-streaming unchanged
- // multi-MB archives on every deploy. HeadObject is a cheap metadata call.
- try {
- const head = await client.send(
- new HeadObjectCommand({ Bucket: bucket, Key: key }),
- );
- if (head.ContentLength === size) {
- skipped++;
- onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`);
- continue;
- }
- } catch {
- // Not found (or HEAD not permitted) → fall through and upload.
- }
- onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`);
- uploaded++;
- const upload = new Upload({
- client,
- params: {
- Bucket: bucket,
- Key: key,
- Body: createReadStream(filePath),
- ContentType: ARCHIVE_CONTENT_TYPE,
- CacheControl: ARCHIVE_CACHE_CONTROL,
- },
- });
- const onAbort = () => {
- upload.abort().catch(() => {});
- };
- signal.addEventListener("abort", onAbort, { once: true });
- try {
- await upload.done();
- } finally {
- signal.removeEventListener("abort", onAbort);
- }
- onLog(`[archives] ${key} done.\n`);
- }
- onLog(
- `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`,
- );
- return 0;
- } catch (err) {
- onLog(
- `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`,
- );
- return 1;
- } finally {
- client.destroy();
- }
-}
-
-// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare
-// Pages project, streaming into `onLog`, returning the exit code. Runs on the
-// host with the host's Cloudflare credentials (process.env) — deploy never runs
-// inside a container, so container wrangler auth is never needed.
-//
-// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to
-// any branch but the project's production branch as a preview, reachable at the
-// branch alias. Omitting it leaves the argv byte-identical to what production
-// has always run — no `--branch`, so wrangler infers the branch from the
-// checkout, which is the long-standing behaviour (and the long-standing hazard:
-// a "production" deploy run from a non-main checkout silently becomes a
-// preview).
-//
-// Either way, the URL wrangler prints ("Take a peek over at …") earns one
-// terminal line of its own, because the streamed log scrolls and an operator
-// who looked away has nowhere else to find it. A preview also gets the stable
-// branch alias, which is knowable without reading the log at all.
-export async function runDeployIntoLog(
- onLog: (line: string) => void,
- signal: AbortSignal,
- site: Site,
- outDir: string,
- paths: Paths,
- opts?: { previewBranch?: string },
-): Promise<number> {
- const project = site.cloudflareProject as string;
- const previewBranch = opts?.previewBranch?.trim() || undefined;
-
- // Spot the deployment URL as it streams past rather than re-reading the
- // finished log file: the log is the operator's too, and buffering it a second
- // time to grep it would double a big deploy's memory for one line of output.
- let deploymentUrl: string | null = null;
- const watch = (line: string) => {
- if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project);
- onLog(line);
- };
-
- const code = await runChildIntoLog(watch, signal, {
- command: "pnpm",
- args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- EXPORT_PUBLIC_DIR: paths.exportPublicDir,
- SITE_ID: site.siteId,
- },
- });
-
- // Only on success. A URL scraped out of a failed run points at nothing — or
- // worse, at the deployment that is still live.
- if (code === 0) {
- if (previewBranch) {
- const alias = previewAliasUrl(project, previewBranch);
- onLog(
- `[preview] ${alias}` +
- (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") +
- "\n",
- );
- } else if (deploymentUrl) {
- onLog(`[deployed] ${deploymentUrl}\n`);
- }
- }
- return code;
-}
-
-// ---------------------------------------------------------------------------
-// Docker export pipeline (buildPipeline.mode = "docker")
-//
-// Three ordered phases (see DEPLOY_DOCKER.md):
-// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives
-// (warm the shared archive cache). One writer of the shared state.
-// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next
-// build in its own container, read-only over the shared caches, writing only
-// its per-site out/ under exportBuildsDir/<siteId>.
-// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase).
-// ---------------------------------------------------------------------------
-
-// The container engine binary. Defaults to `docker`; podman is a CLI drop-in
-// (and rootless podman yields host-owned outputs without needing `-u`).
-function dockerBin(): string {
- return process.env.DOCKER_BIN?.trim() || "docker";
-}
-
-export type SiteBuildOutcome = { siteId: string; code: number };
-export type SiteDeployOutcome = {
- siteId: string;
- status: "deployed" | "skipped" | "failed";
- reason?: string;
-};
-
-// Cheap probe: is the container engine installed and its daemon reachable? Used
-// to fall back to serial host builds when docker isn't available.
-export async function dockerAvailable(signal: AbortSignal): Promise<boolean> {
- const code = await runChildIntoLog(() => {}, signal, {
- command: dockerBin(),
- args: ["version"],
- cwd: process.cwd(),
- });
- return code === 0;
-}
-
-// Run a host-side export pnpm script (build:data / build:archives) for Phase A.
-// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the
-// shared index/staging land at their canonical export/.export-index location
-// (exactly what the fan-out containers mount read-only).
-async function runHostScript(
- onLog: (line: string) => void,
- signal: AbortSignal,
- paths: Paths,
- script: string,
- extraEnv?: Record<string, string>,
-): Promise<number> {
- return runChildIntoLog(onLog, signal, {
- command: "pnpm",
- args: ["run", script],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- ...extraEnv,
- },
- });
-}
-
-// Build (or reuse cached layers of) the per-site build image.
-async function ensureBuildImage(
- onLog: (line: string) => void,
- signal: AbortSignal,
- paths: Paths,
-): Promise<number> {
- const { dockerImage, dockerfile } = getSettings().buildPipeline;
- onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`);
- return runChildIntoLog(onLog, signal, {
- command: dockerBin(),
- args: ["build", "-f", dockerfile, "-t", dockerImage, "."],
- cwd: paths.monorepoRoot,
- });
-}
-
-// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index,
-// and the .export-index staging read-only; mounts the per-site output dir rw.
-// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses
-// next/font/google, which fetches the site's fonts from Google at build time;
-// `--network=none` would fail the build. Isolation still comes from the per-site
-// output dir, the read-only shared mounts, and the non-root `-u` user.
-async function runDockerBuildOne(
- onLog: (line: string) => void,
- signal: AbortSignal,
- siteId: string,
- paths: Paths,
- opts?: { skipArchives?: boolean },
-): Promise<number> {
- const { dockerImage } = getSettings().buildPipeline;
- const siteDir = path.join(paths.exportBuildsDir, siteId);
- // Pre-create the mount target as the host user so container-written files are
- // host-owned (paired with `-u` below), not created root-owned by the daemon.
- await mkdir(siteDir, { recursive: true });
-
- const args = ["run", "--rm", "--init"];
- const uid = typeof process.getuid === "function" ? process.getuid() : null;
- const gid = typeof process.getgid === "function" ? process.getgid() : null;
- if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`);
- // Optional resource caps so N parallel builds (each next build can use ~8 GB)
- // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md.
- const mem = process.env.DOCKER_BUILD_MEMORY?.trim();
- const cpus = process.env.DOCKER_BUILD_CPUS?.trim();
- if (mem) args.push("--memory", mem);
- if (cpus) args.push("--cpus", cpus);
- args.push(
- "-v", `${paths.transcriptsDir}:/data/transcripts:ro`,
- "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`,
- "-v", `${siteDir}:/site`,
- "-e", `SITE_ID=${siteId}`,
- );
- // Mount the host settings.json fresh (build config: archive storage, size caps)
- // rather than relying on a possibly-stale copy — it is NOT baked into the image.
- if (existsSync(paths.settingsFile)) {
- args.push(
- "-v", `${paths.settingsFile}:/data/settings.json:ro`,
- "-e", "SETTINGS_FILE=/data/settings.json",
- );
- }
- // Mount the entrypoint fresh over the baked copy so a script tweak takes effect
- // without an image rebuild (the image still bakes it as a fallback).
- const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh");
- if (existsSync(entrypoint)) {
- args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`);
- }
- if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0");
- args.push(dockerImage);
-
- return runChildIntoLog(onLog, signal, {
- command: dockerBin(),
- args,
- cwd: paths.monorepoRoot,
- label: `[${siteId}] `,
- });
-}
-
-// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an
-// infrastructure failure (data phase / archive warm / image build) that aborts
-// the whole run before any site could build; per-site build failures are
-// returned, not thrown, so one bad site never blocks the rest.
-export async function runDockerBuildAllPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- sites: Site[],
- paths: Paths,
- opts?: { skipArchives?: boolean },
-): Promise<SiteBuildOutcome[]> {
- const { maxParallelBuilds } = getSettings().buildPipeline;
-
- // --- Phase A: shared data + archive cache (host, serial) ---
- onLog("=== Phase A: shared data + archive cache (host) ===");
- const dataCode = await runHostScript(onLog, signal, paths, "build:data");
- if (signal.aborted) return [];
- if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`);
- if (!opts?.skipArchives) {
- const archCode = await runHostScript(onLog, signal, paths, "build:archives");
- if (signal.aborted) return [];
- if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`);
- }
-
- const imgCode = await ensureBuildImage(onLog, signal, paths);
- if (signal.aborted) return [];
- if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`);
-
- // --- Phase B: per-site fan-out (containers, parallel) ---
- onLog(
- `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`,
- );
- return runWithConcurrency(sites, maxParallelBuilds, async (site) => {
- if (signal.aborted) return { siteId: site.siteId, code: 1 };
- const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts);
- onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`);
- return { siteId: site.siteId, code };
- });
-}
-
-// Phase C: deploy each built site SERIALLY on the host, after the build barrier.
-// Partial-failure tolerant — a site that fails to upload/deploy is recorded and
-// the loop continues. Sites that failed to build, or have no Cloudflare project,
-// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site;
-// basic fallback: export/out).
-export async function runDockerDeployAllPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- sites: Site[],
- builtOk: Set<string>,
- paths: Paths,
- outDirFor: (siteId: string) => string,
-): Promise<SiteDeployOutcome[]> {
- const outcomes: SiteDeployOutcome[] = [];
- for (const site of sites) {
- if (signal.aborted) break;
- if (!builtOk.has(site.siteId)) {
- onLog(`[${site.siteId}] deploy skipped — build failed`);
- outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" });
- continue;
- }
- if (!site.cloudflareProject) {
- onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`);
- outcomes.push({
- siteId: site.siteId,
- status: "skipped",
- reason: "no cloudflareProject",
- });
- continue;
- }
- onLog(`=== Deploy ${site.siteId} ===`);
- const uploadCode = await runArchiveUploadIntoLog(
- onLog,
- signal,
- site,
- paths,
- dockerSiteStagingDir(paths, site.siteId),
- );
- if (signal.aborted) break;
- if (uploadCode !== 0) {
- onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`);
- outcomes.push({
- siteId: site.siteId,
- status: "failed",
- reason: `R2 upload exit ${uploadCode}`,
- });
- continue;
- }
- const deployCode = await runDeployIntoLog(
- onLog,
- signal,
- site,
- outDirFor(site.siteId),
- paths,
- );
- if (signal.aborted) break;
- if (deployCode !== 0) {
- onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`);
- outcomes.push({
- siteId: site.siteId,
- status: "failed",
- reason: `deploy exit ${deployCode}`,
- });
- continue;
- }
- onLog(`[${site.siteId}] deployed.`);
- outcomes.push({ siteId: site.siteId, status: "deployed" });
- }
- return outcomes;
-}
-
-// Bounded-concurrency map over a fixed work set, preserving input order in the
-// results. No external dep; a fresh worker pulls the next index until exhausted.
-async function runWithConcurrency<T, R>(
- items: T[],
- limit: number,
- worker: (item: T) => Promise<R>,
-): Promise<R[]> {
- const results: R[] = new Array(items.length);
- let next = 0;
- const width = Math.max(1, Math.min(limit, items.length));
- const runners = Array.from({ length: width }, async () => {
- while (true) {
- const i = next++;
- if (i >= items.length) break;
- results[i] = await worker(items[i]);
- }
- });
- await Promise.all(runners);
- return results;
-}
diff --git a/editor/app/sites/lib/deployAction.ts b/editor/app/sites/lib/deployAction.ts
@@ -13,7 +13,7 @@ import {
resolveOutDir,
runArchiveUploadIntoLog,
runDeployIntoLog,
-} from "./buildDeployCore";
+} from "yt-dlp-transcript-common/publish/build";
const DEPLOY_QUEUE = "deploy";
diff --git a/editor/package.json b/editor/package.json
@@ -14,8 +14,6 @@
"e2e:ui": "playwright test --ui"
},
"dependencies": {
- "@aws-sdk/client-s3": "^3.1080.0",
- "@aws-sdk/lib-storage": "^3.1080.0",
"@sindresorhus/slugify": "^3.0.0",
"lucide-react": "^1.16.0",
"markdown-to-jsx": "^7.7.4",
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
@@ -14,6 +14,12 @@ importers:
common:
dependencies:
+ '@aws-sdk/client-s3':
+ specifier: ^3.1080.0
+ version: 3.1080.0
+ '@aws-sdk/lib-storage':
+ specifier: ^3.1080.0
+ version: 3.1080.0(@aws-sdk/client-s3@3.1080.0)
'@sindresorhus/slugify':
specifier: ^3.0.0
version: 3.0.0
@@ -114,12 +120,6 @@ importers:
editor:
dependencies:
- '@aws-sdk/client-s3':
- specifier: ^3.1080.0
- version: 3.1080.0
- '@aws-sdk/lib-storage':
- specifier: ^3.1080.0
- version: 3.1080.0(@aws-sdk/client-s3@3.1080.0)
'@sindresorhus/slugify':
specifier: ^3.0.0
version: 3.0.0