commit 1e486c64d1ffcd52a9c8a4f2e9bf7ef731640a64
parent 0e9d5dd08642f0b334b4f3ab6aaf629960d67bfa
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 01:02:38 -0400
Merge one-core/phase-4-s1 — Phase 4 slice 1: buildDeployCore moves to common/publish/build.ts, aws-sdk deps move to common
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
13 files changed, 714 insertions(+), 598 deletions(-)
diff --git a/DEPLOY_CLOUDFLARE.md b/DEPLOY_CLOUDFLARE.md
@@ -242,7 +242,7 @@ small and commented if you want to adjust them.
The archive `Cache-Control` is also set at upload time (`Cache-Control: public,
max-age=3600`, constant `ARCHIVE_CACHE_CONTROL` in
-`editor/app/deploy/buildDeployCore.ts`); the Worker reads it back when caching. Archive
+`common/publish/build.ts`); the Worker reads it back when caching. Archive
filenames are stable and overwritten in place on re-deploy, so this 1-hour bound is what
keeps a re-uploaded archive from being served stale for long — raise it if your archives
rarely change.
diff --git a/common/architecture.test.ts b/common/architecture.test.ts
@@ -35,15 +35,20 @@ const FORBIDDEN: Record<string, readonly string[]> = {
// searchQuery for a mode union. All four inverted — the pipeline and the two
// types moved down, and the fetch and the memo are now injected — so the
// guard turns on with nothing added to the allow-list to pay for it.
- lib: ["controller", "jobs", "components", "views"],
- jobs: ["controller", "views"],
- controller: ["views"],
- components: ["controller", "jobs", "ytdlp"],
+ lib: ["controller", "jobs", "components", "views", "publish"],
+ jobs: ["controller", "views", "publish"],
+ controller: ["views", "publish"],
+ components: ["controller", "jobs", "ytdlp", "publish"],
// `views/` is the view-model layer (one-core phase 3 slice 1): pure functions
// that fold live state into a payload. It sits ABOVE dispatch and BELOW ui,
// so it may read lib/, controller/ and jobs/ for pure helpers and types, and
// may not reach up into components/ or sideways into the entry points.
views: ["components", "ytdlp", "bin", "social"],
+ // `publish/` is the publish layer (one-core phase 4 slice 1): building a site,
+ // uploading its archives, deploying it. It sits above dispatch and below views,
+ // so it may read lib/, jobs/ and controller/, and may not reach up into
+ // views/ or components/.
+ publish: ["views", "components"],
};
// The directories walked. `bin/` and `social/` are scanned so a back-edge cannot
@@ -57,6 +62,7 @@ const ROOTS = [
"ytdlp",
"bin",
"views",
+ "publish",
];
// Today's back-edges, `<file> -> <imported module>`, each with why it is still
@@ -164,10 +170,10 @@ test("no new back-edges between common's layers", async () => {
assert.deepEqual(
unexpected,
[],
- `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, ` +
+ `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, publish/, ` +
`components/ or views/; ` +
- `jobs/ and controller/ may not import views/; components/ may not ` +
- `import controller/, jobs/ or ytdlp/; views/ may not import ` +
+ `jobs/ and controller/ may not import views/ or publish/; components/ may not ` +
+ `import controller/, jobs/, ytdlp/ or publish/; publish/ may not import views/ or components/; views/ may not import ` +
`components/. Move the type or the function down a ` +
`layer instead of adding it to ALLOWED. Found: ${unexpected.join(", ")}`,
);
diff --git a/common/package.json b/common/package.json
@@ -37,6 +37,7 @@
"./controller/*": "./controller/*.ts",
"./jobs/*": "./jobs/*.ts",
"./views/*": "./views/*.ts",
+ "./publish/*": "./publish/*.ts",
"./social/*": "./social/*.ts",
"./ytdlp/*": "./ytdlp/*.ts",
"./bin/*": "./bin/*.ts",
@@ -44,9 +45,11 @@
"./styles/*": "./styles/*.ts"
},
"scripts": {
- "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*/*.test.ts\""
+ "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*/*.test.ts\""
},
"dependencies": {
+ "@aws-sdk/client-s3": "^3.1080.0",
+ "@aws-sdk/lib-storage": "^3.1080.0",
"@sindresorhus/slugify": "^3.0.0",
"@tanstack/react-query": "^5.99.1",
"@tanstack/react-virtual": "^3.13.0",
diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts
@@ -0,0 +1,43 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Paths } from "../lib/paths";
+import {
+ dockerSiteOutDir,
+ dockerSiteStagingDir,
+ resolveOutDir,
+} from "./build";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// The three pure path helpers of the publish layer. Their outputs are the
+// on-disk contract the build and deploy phases share (and that
+// `docker/build-site.sh` mounts), so they are pinned here exactly.
+
+const paths = {
+ exportDir: "/repo/export",
+ exportBuildsDir: "/repo/export/.export-builds",
+} as Paths;
+
+test("resolveOutDir is export/out whatever the site", () => {
+ assert.equal(resolveOutDir("jeralyzer", paths), "/repo/export/out");
+ assert.equal(resolveOutDir("anilyzer", paths), "/repo/export/out");
+});
+
+test("dockerSiteOutDir is a per-site out/ under exportBuildsDir", () => {
+ assert.equal(
+ dockerSiteOutDir(paths, "jeralyzer"),
+ "/repo/export/.export-builds/jeralyzer/out",
+ );
+ assert.notEqual(
+ dockerSiteOutDir(paths, "jeralyzer"),
+ dockerSiteOutDir(paths, "anilyzer"),
+ );
+});
+
+test("dockerSiteStagingDir nests .r2-staging/<site>/archives under the site's build dir", () => {
+ assert.equal(
+ dockerSiteStagingDir(paths, "jeralyzer"),
+ "/repo/export/.export-builds/jeralyzer/.r2-staging/jeralyzer/archives",
+ );
+});
diff --git a/common/publish/build.ts b/common/publish/build.ts
@@ -0,0 +1,582 @@
+// The publish layer's build/deploy primitives: one site's host build, the
+// docker per-site fan-out, the R2 archive upload and the Pages deploy. Moved
+// here from the editor (`editor/app/sites/lib/buildDeployCore.ts`) in one-core
+// Phase 4 slice 1, unchanged; the editor's build/deploy server actions
+// (`editor/app/sites/lib/{buildAction,deployAction}.ts`) call them as jobs.
+//
+// These take an `onLog` callback and an AbortSignal, so they CANNOT live in a
+// "use server" module (every export there becomes a server action, which
+// forbids non-serializable args). Keep them as plain helpers.
+
+import path from "node:path";
+import { mkdir, readdir, stat } from "node:fs/promises";
+import { createReadStream, existsSync } from "node:fs";
+import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
+import { Upload } from "@aws-sdk/lib-storage";
+import { runChildIntoLog } from "../jobs/runChild";
+import {
+ deploymentUrlIn,
+ pagesDeployArgs,
+ previewAliasUrl,
+} from "../lib/pagesDeploy";
+import type { Paths } from "../lib/paths";
+import { getSettings } from "../lib/settings";
+import type { Site } from "../lib/site";
+
+// Where the basic (host) build writes the static bundle to deploy: the fixed
+// export/out, composed one site at a time. The docker fan-out writes per-site
+// out/ dirs instead — see dockerSiteOutDir.
+export function resolveOutDir(_siteId: string, paths: Paths): string {
+ return path.join(paths.exportDir, "out");
+}
+
+// Where a docker per-site container writes its built bundle (mounted as /site/out
+// inside the container). Isolated per site so parallel builds never collide.
+export function dockerSiteOutDir(paths: Paths, siteId: string): string {
+ return path.join(paths.exportBuildsDir, siteId, "out");
+}
+
+// Where a docker per-site container stages oversize archives for R2 upload. The
+// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/
+// public), so compose writes .r2-staging as its sibling under exportBuildsDir.
+export function dockerSiteStagingDir(paths: Paths, siteId: string): string {
+ return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives");
+}
+
+// Run the basic (host) build phase for one site, streaming into `onLog`,
+// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream
+// on the build queue, since the export/ tree is shared). This is the single-site
+// build for basic mode, and the fallback the all-sites docker action drops to
+// when no container engine is available. The parallel per-site container path
+// lives in runDockerBuildAllPhase.
+export async function runBuildPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ siteId: string,
+ paths: Paths,
+ opts?: { skipData?: boolean; skipArchives?: boolean },
+): Promise<number> {
+ // When skipping the data rebuild, run the `build:nodata` script instead of
+ // `build`. `build:nodata` has the same body (compose:site + next build) but a
+ // different name, so npm's `prebuild` lifecycle hook (which runs build:data =
+ // index/stats/templates) does NOT fire — we compose from the existing
+ // .export-index staging. Assumes a prior full build produced that staging.
+ const skipData = opts?.skipData === true;
+ if (skipData) {
+ onLog(
+ "[notice] Skipping data rebuild (index/stats/charts) — composing from " +
+ "existing .export-index staging.\n",
+ );
+ }
+ // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes
+ // compose-site skip generation this build regardless of the global/site flags.
+ const skipArchives = opts?.skipArchives === true;
+ if (skipArchives) {
+ onLog("[notice] Skipping archive-zip generation for this build.\n");
+ }
+ return runChildIntoLog(onLog, signal, {
+ command: "pnpm",
+ args: ["run", skipData ? "build:nodata" : "build"],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ EXPORT_PUBLIC_DIR: paths.exportPublicDir,
+ SITE_ID: siteId,
+ ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}),
+ },
+ });
+}
+
+// Where the basic (host) compose staged this site's oversize archives for R2
+// upload. Kept outside export/public so they never ship as Pages assets. Mirrors
+// the path composeArchives writes to in common/bin/compose-site.ts. The docker
+// fan-out uses dockerSiteStagingDir instead, passed explicitly.
+function archiveStagingDir(siteId: string, paths: Paths): string {
+ return path.join(
+ path.dirname(paths.exportPublicDir),
+ ".r2-staging",
+ siteId,
+ "archives",
+ );
+}
+
+// Said once at the top of every preview deploy, because the one thing a preview
+// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a
+// preview's oversize archives overwrite the keys production's manifest points
+// at. That is cheap and harmless in practice — the upload skips any object R2
+// already holds at the same size, and an unchanged channel re-zips byte-stable
+// — but "harmless because of a size check" is exactly the kind of thing an
+// operator should be told rather than left to discover.
+export const PREVIEW_SHARES_ARCHIVES_NOTICE =
+ "[notice] A preview shares the production R2 archive bucket — unchanged " +
+ "archives are skipped, so this is cheap, but a CHANGED archive replaces the " +
+ "one production links to.\n";
+
+// Cache-Control set on every uploaded archive. Served through a Cloudflare custom
+// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead
+// of hitting R2 (each origin GET is a billable Class B op), which is the main cost
+// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood
+// absorption against re-deployed archives (stable filenames, overwritten in place)
+// going stale; raise it if your archives rarely change.
+const ARCHIVE_CACHE_CONTROL = "public, max-age=3600";
+
+// MIME type stored on each uploaded archive. Archives are `.zip` bundles.
+const ARCHIVE_CONTENT_TYPE = "application/zip";
+
+// Upload this site's staged oversize archives to the configured R2 bucket, so the
+// remote URLs the served manifest points at actually resolve. Uploads go through
+// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) —
+// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real
+// live-chat archives blow past, whereas multipart streams any size. Keys match
+// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must
+// run BEFORE the Pages deploy so the manifest never points at a missing object.
+//
+// No-op (returns 0) when overflow storage isn't configured or nothing was staged.
+// Credentials come from the host env (NOT the settings JSON — secrets don't
+// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID
+// (for the S3 endpoint). If a bucket is configured and files are staged but the
+// credentials are missing, we fail (return non-zero) so the deploy aborts rather
+// than shipping a manifest that points at un-uploaded objects.
+export async function runArchiveUploadIntoLog(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ site: Site,
+ paths: Paths,
+ // Where the oversize archives were staged. Defaults to the basic host location;
+ // the docker deploy phase passes dockerSiteStagingDir since its container wrote
+ // .r2-staging under exportBuildsDir/<siteId> instead.
+ stagingDirOverride?: string,
+): Promise<number> {
+ const bucket = getSettings().archiveStorage?.bucket?.trim();
+ if (!bucket) return 0;
+
+ const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths);
+ let files: string[];
+ try {
+ files = (await readdir(stagingDir)).filter((f) => !f.startsWith("."));
+ } catch {
+ // No staging dir → nothing oversize this build.
+ return 0;
+ }
+ if (files.length === 0) return 0;
+
+ const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim();
+ const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim();
+ const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim();
+ if (!accessKeyId || !secretAccessKey || !accountId) {
+ onLog(
+ `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` +
+ `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` +
+ `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` +
+ `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` +
+ `missing files.\n`,
+ );
+ return 1;
+ }
+
+ const client = new S3Client({
+ region: "auto",
+ endpoint: `https://${accountId}.r2.cloudflarestorage.com`,
+ credentials: { accessKeyId, secretAccessKey },
+ });
+
+ onLog(
+ `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`,
+ );
+ try {
+ let uploaded = 0;
+ let skipped = 0;
+ for (const file of files) {
+ if (signal.aborted) return 1;
+ const key = `${site.siteId}/archives/${file}`;
+ const filePath = path.join(stagingDir, file);
+ const { size } = await stat(filePath);
+ // Skip the re-upload when R2 already holds an object of the same size for
+ // this key. The archive cache makes an unchanged channel's zip byte-stable
+ // build-to-build, so a same-size object is the same object; a changed
+ // channel re-zips to a different size. Avoids re-streaming unchanged
+ // multi-MB archives on every deploy. HeadObject is a cheap metadata call.
+ try {
+ const head = await client.send(
+ new HeadObjectCommand({ Bucket: bucket, Key: key }),
+ );
+ if (head.ContentLength === size) {
+ skipped++;
+ onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`);
+ continue;
+ }
+ } catch {
+ // Not found (or HEAD not permitted) → fall through and upload.
+ }
+ onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`);
+ uploaded++;
+ const upload = new Upload({
+ client,
+ params: {
+ Bucket: bucket,
+ Key: key,
+ Body: createReadStream(filePath),
+ ContentType: ARCHIVE_CONTENT_TYPE,
+ CacheControl: ARCHIVE_CACHE_CONTROL,
+ },
+ });
+ const onAbort = () => {
+ upload.abort().catch(() => {});
+ };
+ signal.addEventListener("abort", onAbort, { once: true });
+ try {
+ await upload.done();
+ } finally {
+ signal.removeEventListener("abort", onAbort);
+ }
+ onLog(`[archives] ${key} done.\n`);
+ }
+ onLog(
+ `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`,
+ );
+ return 0;
+ } catch (err) {
+ onLog(
+ `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`,
+ );
+ return 1;
+ } finally {
+ client.destroy();
+ }
+}
+
+// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare
+// Pages project, streaming into `onLog`, returning the exit code. Runs on the
+// host with the host's Cloudflare credentials (process.env) — deploy never runs
+// inside a container, so container wrangler auth is never needed.
+//
+// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to
+// any branch but the project's production branch as a preview, reachable at the
+// branch alias. Omitting it leaves the argv byte-identical to what production
+// has always run — no `--branch`, so wrangler infers the branch from the
+// checkout, which is the long-standing behaviour (and the long-standing hazard:
+// a "production" deploy run from a non-main checkout silently becomes a
+// preview).
+//
+// Either way, the URL wrangler prints ("Take a peek over at …") earns one
+// terminal line of its own, because the streamed log scrolls and an operator
+// who looked away has nowhere else to find it. A preview also gets the stable
+// branch alias, which is knowable without reading the log at all.
+export async function runDeployIntoLog(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ site: Site,
+ outDir: string,
+ paths: Paths,
+ opts?: { previewBranch?: string },
+): Promise<number> {
+ const project = site.cloudflareProject as string;
+ const previewBranch = opts?.previewBranch?.trim() || undefined;
+
+ // Spot the deployment URL as it streams past rather than re-reading the
+ // finished log file: the log is the operator's too, and buffering it a second
+ // time to grep it would double a big deploy's memory for one line of output.
+ let deploymentUrl: string | null = null;
+ const watch = (line: string) => {
+ if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project);
+ onLog(line);
+ };
+
+ const code = await runChildIntoLog(watch, signal, {
+ command: "pnpm",
+ args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ EXPORT_PUBLIC_DIR: paths.exportPublicDir,
+ SITE_ID: site.siteId,
+ },
+ });
+
+ // Only on success. A URL scraped out of a failed run points at nothing — or
+ // worse, at the deployment that is still live.
+ if (code === 0) {
+ if (previewBranch) {
+ const alias = previewAliasUrl(project, previewBranch);
+ onLog(
+ `[preview] ${alias}` +
+ (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") +
+ "\n",
+ );
+ } else if (deploymentUrl) {
+ onLog(`[deployed] ${deploymentUrl}\n`);
+ }
+ }
+ return code;
+}
+
+// ---------------------------------------------------------------------------
+// Docker export pipeline (buildPipeline.mode = "docker")
+//
+// Three ordered phases (see DEPLOY_DOCKER.md):
+// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives
+// (warm the shared archive cache). One writer of the shared state.
+// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next
+// build in its own container, read-only over the shared caches, writing only
+// its per-site out/ under exportBuildsDir/<siteId>.
+// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase).
+// ---------------------------------------------------------------------------
+
+// The container engine binary. Defaults to `docker`; podman is a CLI drop-in
+// (and rootless podman yields host-owned outputs without needing `-u`).
+function dockerBin(): string {
+ return process.env.DOCKER_BIN?.trim() || "docker";
+}
+
+export type SiteBuildOutcome = { siteId: string; code: number };
+export type SiteDeployOutcome = {
+ siteId: string;
+ status: "deployed" | "skipped" | "failed";
+ reason?: string;
+};
+
+// Cheap probe: is the container engine installed and its daemon reachable? Used
+// to fall back to serial host builds when docker isn't available.
+export async function dockerAvailable(signal: AbortSignal): Promise<boolean> {
+ const code = await runChildIntoLog(() => {}, signal, {
+ command: dockerBin(),
+ args: ["version"],
+ cwd: process.cwd(),
+ });
+ return code === 0;
+}
+
+// Run a host-side export pnpm script (build:data / build:archives) for Phase A.
+// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the
+// shared index/staging land at their canonical export/.export-index location
+// (exactly what the fan-out containers mount read-only).
+async function runHostScript(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ paths: Paths,
+ script: string,
+ extraEnv?: Record<string, string>,
+): Promise<number> {
+ return runChildIntoLog(onLog, signal, {
+ command: "pnpm",
+ args: ["run", script],
+ cwd: paths.exportDir,
+ env: {
+ ...process.env,
+ NODE_ENV: "production",
+ TRANSCRIPTS_DIR: paths.transcriptsDir,
+ ...extraEnv,
+ },
+ });
+}
+
+// Build (or reuse cached layers of) the per-site build image.
+async function ensureBuildImage(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ paths: Paths,
+): Promise<number> {
+ const { dockerImage, dockerfile } = getSettings().buildPipeline;
+ onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`);
+ return runChildIntoLog(onLog, signal, {
+ command: dockerBin(),
+ args: ["build", "-f", dockerfile, "-t", dockerImage, "."],
+ cwd: paths.monorepoRoot,
+ });
+}
+
+// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index,
+// and the .export-index staging read-only; mounts the per-site output dir rw.
+// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses
+// next/font/google, which fetches the site's fonts from Google at build time;
+// `--network=none` would fail the build. Isolation still comes from the per-site
+// output dir, the read-only shared mounts, and the non-root `-u` user.
+async function runDockerBuildOne(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ siteId: string,
+ paths: Paths,
+ opts?: { skipArchives?: boolean },
+): Promise<number> {
+ const { dockerImage } = getSettings().buildPipeline;
+ const siteDir = path.join(paths.exportBuildsDir, siteId);
+ // Pre-create the mount target as the host user so container-written files are
+ // host-owned (paired with `-u` below), not created root-owned by the daemon.
+ await mkdir(siteDir, { recursive: true });
+
+ const args = ["run", "--rm", "--init"];
+ const uid = typeof process.getuid === "function" ? process.getuid() : null;
+ const gid = typeof process.getgid === "function" ? process.getgid() : null;
+ if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`);
+ // Optional resource caps so N parallel builds (each next build can use ~8 GB)
+ // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md.
+ const mem = process.env.DOCKER_BUILD_MEMORY?.trim();
+ const cpus = process.env.DOCKER_BUILD_CPUS?.trim();
+ if (mem) args.push("--memory", mem);
+ if (cpus) args.push("--cpus", cpus);
+ args.push(
+ "-v", `${paths.transcriptsDir}:/data/transcripts:ro`,
+ "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`,
+ "-v", `${siteDir}:/site`,
+ "-e", `SITE_ID=${siteId}`,
+ );
+ // Mount the host settings.json fresh (build config: archive storage, size caps)
+ // rather than relying on a possibly-stale copy — it is NOT baked into the image.
+ if (existsSync(paths.settingsFile)) {
+ args.push(
+ "-v", `${paths.settingsFile}:/data/settings.json:ro`,
+ "-e", "SETTINGS_FILE=/data/settings.json",
+ );
+ }
+ // Mount the entrypoint fresh over the baked copy so a script tweak takes effect
+ // without an image rebuild (the image still bakes it as a fallback).
+ const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh");
+ if (existsSync(entrypoint)) {
+ args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`);
+ }
+ if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0");
+ args.push(dockerImage);
+
+ return runChildIntoLog(onLog, signal, {
+ command: dockerBin(),
+ args,
+ cwd: paths.monorepoRoot,
+ label: `[${siteId}] `,
+ });
+}
+
+// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an
+// infrastructure failure (data phase / archive warm / image build) that aborts
+// the whole run before any site could build; per-site build failures are
+// returned, not thrown, so one bad site never blocks the rest.
+export async function runDockerBuildAllPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ sites: Site[],
+ paths: Paths,
+ opts?: { skipArchives?: boolean },
+): Promise<SiteBuildOutcome[]> {
+ const { maxParallelBuilds } = getSettings().buildPipeline;
+
+ // --- Phase A: shared data + archive cache (host, serial) ---
+ onLog("=== Phase A: shared data + archive cache (host) ===");
+ const dataCode = await runHostScript(onLog, signal, paths, "build:data");
+ if (signal.aborted) return [];
+ if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`);
+ if (!opts?.skipArchives) {
+ const archCode = await runHostScript(onLog, signal, paths, "build:archives");
+ if (signal.aborted) return [];
+ if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`);
+ }
+
+ const imgCode = await ensureBuildImage(onLog, signal, paths);
+ if (signal.aborted) return [];
+ if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`);
+
+ // --- Phase B: per-site fan-out (containers, parallel) ---
+ onLog(
+ `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`,
+ );
+ return runWithConcurrency(sites, maxParallelBuilds, async (site) => {
+ if (signal.aborted) return { siteId: site.siteId, code: 1 };
+ const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts);
+ onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`);
+ return { siteId: site.siteId, code };
+ });
+}
+
+// Phase C: deploy each built site SERIALLY on the host, after the build barrier.
+// Partial-failure tolerant — a site that fails to upload/deploy is recorded and
+// the loop continues. Sites that failed to build, or have no Cloudflare project,
+// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site;
+// basic fallback: export/out).
+export async function runDockerDeployAllPhase(
+ onLog: (line: string) => void,
+ signal: AbortSignal,
+ sites: Site[],
+ builtOk: Set<string>,
+ paths: Paths,
+ outDirFor: (siteId: string) => string,
+): Promise<SiteDeployOutcome[]> {
+ const outcomes: SiteDeployOutcome[] = [];
+ for (const site of sites) {
+ if (signal.aborted) break;
+ if (!builtOk.has(site.siteId)) {
+ onLog(`[${site.siteId}] deploy skipped — build failed`);
+ outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" });
+ continue;
+ }
+ if (!site.cloudflareProject) {
+ onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "skipped",
+ reason: "no cloudflareProject",
+ });
+ continue;
+ }
+ onLog(`=== Deploy ${site.siteId} ===`);
+ const uploadCode = await runArchiveUploadIntoLog(
+ onLog,
+ signal,
+ site,
+ paths,
+ dockerSiteStagingDir(paths, site.siteId),
+ );
+ if (signal.aborted) break;
+ if (uploadCode !== 0) {
+ onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "failed",
+ reason: `R2 upload exit ${uploadCode}`,
+ });
+ continue;
+ }
+ const deployCode = await runDeployIntoLog(
+ onLog,
+ signal,
+ site,
+ outDirFor(site.siteId),
+ paths,
+ );
+ if (signal.aborted) break;
+ if (deployCode !== 0) {
+ onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`);
+ outcomes.push({
+ siteId: site.siteId,
+ status: "failed",
+ reason: `deploy exit ${deployCode}`,
+ });
+ continue;
+ }
+ onLog(`[${site.siteId}] deployed.`);
+ outcomes.push({ siteId: site.siteId, status: "deployed" });
+ }
+ return outcomes;
+}
+
+// Bounded-concurrency map over a fixed work set, preserving input order in the
+// results. No external dep; a fresh worker pulls the next index until exhausted.
+async function runWithConcurrency<T, R>(
+ items: T[],
+ limit: number,
+ worker: (item: T) => Promise<R>,
+): Promise<R[]> {
+ const results: R[] = new Array(items.length);
+ let next = 0;
+ const width = Math.max(1, Math.min(limit, items.length));
+ const runners = Array.from({ length: width }, async () => {
+ while (true) {
+ const i = next++;
+ if (i >= items.length) break;
+ results[i] = await worker(items[i]);
+ }
+ });
+ await Promise.all(runners);
+ return results;
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -4,6 +4,7 @@
- **A video the server answers with HTTP 410 Gone is recorded as removed, not as an error.** Rumble answers a taken-down video with `HTTP Error 410: Gone`; the availability check read that as a generic error (one Rekieta Law Rumble video has said "error" since 2026-08-21), and a download that hit it could stop the batch. It now reads as removed, like "Video unavailable" does, so the check says so and a download skips that one video and carries on. Existing records change the next time the video is checked.
- **umtool's report videos can fetch Rumble clips again.** The clip fetch and the source availability check in `umtool/report-to-video` ran yt-dlp without the browser fingerprint Rumble now requires, so every Rumble clip failed with 403 and every Rumble source looked missing. They now pass the same Rumble arguments as the editor, from the same single table.
- **A transcript pulled back from a remote worker is written safely.** It used to be written straight onto `transcript.json`, so a crash part-way through left a truncated transcript; it now goes through the editor's one atomic write (temp file, then rename), like every other file the editor writes.
+- **Nothing changes when you build or deploy a site; the code that does it has moved into the shared core.** The site build, the docker per-site fan-out, the R2 archive upload and the Cloudflare Pages deploy used to live inside the editor. They are now `common/publish/build.ts`, with the same log lines, exit codes and output paths, so a later command-line tool can build and deploy without the editor. The editor's Build, Deploy and Build & deploy controls and `pnpm ops build-site` / `build-deploy` / `deploy-site` call them as before. The AWS SDK packages used for the R2 upload moved with the code, from the editor's dependencies to the core's.
- **Rumble works again, and a Rumble full sweep that gets rate-limited no longer fails the sync.** Every Rumble request had started coming back 403 from Cloudflare unless yt-dlp presents a browser fingerprint (yt-dlp #17496), so Rumble downloads failed and a Rumble channel could not even be added. Every yt-dlp run for a Rumble channel — sync, download, metadata scan, availability check, the clip-window fetch and the new-channel probe — now passes `--impersonate chrome --sleep-requests 1`, from one table in the code; a channel's own extra yt-dlp arguments still come last and still win. Separately, a full sweep that hits HTTP 429 part-way through the listing used to fail the whole sync and try again on the next one, so a large channel (The Quartering on Rumble, 44 days) never synced at all. What it read is now treated as *incomplete* — not a listing, so nothing is flagged missing and the stored playlist is untouched: the job records the platform's rate-limit cooldown, says "sweep incomplete: 429 at page N of the listing, M entries" in its log, does the ordinary newest-first sync instead, and succeeds. Syncs for that platform are then refused until its cooldown ends, and the full sweep is tried again after that. Any other yt-dlp failure still fails the sync as before.
- **A site can turn off its visitors' per-video transcript downloads.** The transcript viewer on a published site has always offered three ways to take a video's text away: a **Download** menu (txt, srt, json), **Copy MD**, and **Copy download command** (a `yt-dlp` line for a marked clip). A site's settings form now has a checkbox for them, *Per-video transcript downloads*, beside the archive zips one. Unticked, the site's next build shows none of the three; **Share** and the clip marks stay. It is on by default, so a site nobody touches is unchanged, and the file stores `"transcriptDownloads": false` only when it is off (`SITE.md` has the key). The site's machine contract (`/corpus.json`, `llms.txt`, the manifests and shards the MCP server and report-to-video read) is published either way. The hub follows the same switch: the hub form on **Sites** has the same checkbox, stored as `"transcriptDownloads": false` in the hub's `homepage.json`, and it hides the three controls on the hub's Browse and Ask pages. The editor's own video pages are unaffected.
- **Channel rows no longer scroll over a group's controls on `/channels`.** Scrolled down and to the right, the pinned Slug column of every row painted over the pinned group header and its five station buttons (Sync, Download, Transcribe, Digest and the speaker lane), and took the clicks. The pinned Slug cell and the group header sat at the same stacking level, and the later rows won. The rack now has one named layer order, kept in one file: the Advanced panel, then the column header, then the group header, then the pinned checkbox and Slug cells. Nothing ties any more. The screenshot audit found four more problems, fixed as well. A group header's name and buttons now stay on screen however far the columns scroll across (they used to scroll off to the left). An Advanced panel opened near the bottom or the right edge scrolls itself into view instead of being cut off. The rule above a pinned group header moves with it instead of leaving a gap the rows showed through. On a phone, the column header no longer paints over the selection bar pinned to the bottom of the screen.
diff --git a/editor/app/sites/lib/buildAction.ts b/editor/app/sites/lib/buildAction.ts
@@ -33,7 +33,7 @@ import {
runDockerDeployAllPhase,
type SiteBuildOutcome,
type SiteDeployOutcome,
-} from "./buildDeployCore";
+} from "yt-dlp-transcript-common/publish/build";
const DEFAULT_BUILD_QUEUE = "build";
const DEPLOY_QUEUE = "deploy";
diff --git a/editor/app/sites/lib/buildDeployCore.ts b/editor/app/sites/lib/buildDeployCore.ts
@@ -1,578 +0,0 @@
-// Shared, server-only build/deploy primitives used by the build/deploy server
-// actions (buildAction.ts, deployAction.ts). These take an `onLog` callback and
-// an AbortSignal, so they CANNOT live in a "use server" module (every export
-// there becomes a server action, which forbids non-serializable args). Keep them
-// here as plain helpers and import them into the thin action wrappers.
-
-import path from "node:path";
-import { mkdir, readdir, stat } from "node:fs/promises";
-import { createReadStream, existsSync } from "node:fs";
-import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3";
-import { Upload } from "@aws-sdk/lib-storage";
-import { runChildIntoLog } from "yt-dlp-transcript-common/jobs/runChild";
-import {
- deploymentUrlIn,
- pagesDeployArgs,
- previewAliasUrl,
-} from "yt-dlp-transcript-common/lib/pagesDeploy";
-import type { Paths } from "yt-dlp-transcript-common/lib/paths";
-import { getSettings } from "yt-dlp-transcript-common/lib/settings";
-import type { Site } from "yt-dlp-transcript-common/lib/site";
-
-// Where the basic (host) build writes the static bundle to deploy: the fixed
-// export/out, composed one site at a time. The docker fan-out writes per-site
-// out/ dirs instead — see dockerSiteOutDir.
-export function resolveOutDir(_siteId: string, paths: Paths): string {
- return path.join(paths.exportDir, "out");
-}
-
-// Where a docker per-site container writes its built bundle (mounted as /site/out
-// inside the container). Isolated per site so parallel builds never collide.
-export function dockerSiteOutDir(paths: Paths, siteId: string): string {
- return path.join(paths.exportBuildsDir, siteId, "out");
-}
-
-// Where a docker per-site container stages oversize archives for R2 upload. The
-// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/
-// public), so compose writes .r2-staging as its sibling under exportBuildsDir.
-export function dockerSiteStagingDir(paths: Paths, siteId: string): string {
- return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives");
-}
-
-// Run the basic (host) build phase for one site, streaming into `onLog`,
-// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream
-// on the build queue, since the export/ tree is shared). This is the single-site
-// build for basic mode, and the fallback the all-sites docker action drops to
-// when no container engine is available. The parallel per-site container path
-// lives in runDockerBuildAllPhase.
-export async function runBuildPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- siteId: string,
- paths: Paths,
- opts?: { skipData?: boolean; skipArchives?: boolean },
-): Promise<number> {
- // When skipping the data rebuild, run the `build:nodata` script instead of
- // `build`. `build:nodata` has the same body (compose:site + next build) but a
- // different name, so npm's `prebuild` lifecycle hook (which runs build:data =
- // index/stats/templates) does NOT fire — we compose from the existing
- // .export-index staging. Assumes a prior full build produced that staging.
- const skipData = opts?.skipData === true;
- if (skipData) {
- onLog(
- "[notice] Skipping data rebuild (index/stats/charts) — composing from " +
- "existing .export-index staging.\n",
- );
- }
- // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes
- // compose-site skip generation this build regardless of the global/site flags.
- const skipArchives = opts?.skipArchives === true;
- if (skipArchives) {
- onLog("[notice] Skipping archive-zip generation for this build.\n");
- }
- return runChildIntoLog(onLog, signal, {
- command: "pnpm",
- args: ["run", skipData ? "build:nodata" : "build"],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- EXPORT_PUBLIC_DIR: paths.exportPublicDir,
- SITE_ID: siteId,
- ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}),
- },
- });
-}
-
-// Where the basic (host) compose staged this site's oversize archives for R2
-// upload. Kept outside export/public so they never ship as Pages assets. Mirrors
-// the path composeArchives writes to in common/bin/compose-site.ts. The docker
-// fan-out uses dockerSiteStagingDir instead, passed explicitly.
-function archiveStagingDir(siteId: string, paths: Paths): string {
- return path.join(
- path.dirname(paths.exportPublicDir),
- ".r2-staging",
- siteId,
- "archives",
- );
-}
-
-// Said once at the top of every preview deploy, because the one thing a preview
-// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a
-// preview's oversize archives overwrite the keys production's manifest points
-// at. That is cheap and harmless in practice — the upload skips any object R2
-// already holds at the same size, and an unchanged channel re-zips byte-stable
-// — but "harmless because of a size check" is exactly the kind of thing an
-// operator should be told rather than left to discover.
-export const PREVIEW_SHARES_ARCHIVES_NOTICE =
- "[notice] A preview shares the production R2 archive bucket — unchanged " +
- "archives are skipped, so this is cheap, but a CHANGED archive replaces the " +
- "one production links to.\n";
-
-// Cache-Control set on every uploaded archive. Served through a Cloudflare custom
-// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead
-// of hitting R2 (each origin GET is a billable Class B op), which is the main cost
-// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood
-// absorption against re-deployed archives (stable filenames, overwritten in place)
-// going stale; raise it if your archives rarely change.
-const ARCHIVE_CACHE_CONTROL = "public, max-age=3600";
-
-// MIME type stored on each uploaded archive. Archives are `.zip` bundles.
-const ARCHIVE_CONTENT_TYPE = "application/zip";
-
-// Upload this site's staged oversize archives to the configured R2 bucket, so the
-// remote URLs the served manifest points at actually resolve. Uploads go through
-// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) —
-// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real
-// live-chat archives blow past, whereas multipart streams any size. Keys match
-// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must
-// run BEFORE the Pages deploy so the manifest never points at a missing object.
-//
-// No-op (returns 0) when overflow storage isn't configured or nothing was staged.
-// Credentials come from the host env (NOT the settings JSON — secrets don't
-// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID
-// (for the S3 endpoint). If a bucket is configured and files are staged but the
-// credentials are missing, we fail (return non-zero) so the deploy aborts rather
-// than shipping a manifest that points at un-uploaded objects.
-export async function runArchiveUploadIntoLog(
- onLog: (line: string) => void,
- signal: AbortSignal,
- site: Site,
- paths: Paths,
- // Where the oversize archives were staged. Defaults to the basic host location;
- // the docker deploy phase passes dockerSiteStagingDir since its container wrote
- // .r2-staging under exportBuildsDir/<siteId> instead.
- stagingDirOverride?: string,
-): Promise<number> {
- const bucket = getSettings().archiveStorage?.bucket?.trim();
- if (!bucket) return 0;
-
- const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths);
- let files: string[];
- try {
- files = (await readdir(stagingDir)).filter((f) => !f.startsWith("."));
- } catch {
- // No staging dir → nothing oversize this build.
- return 0;
- }
- if (files.length === 0) return 0;
-
- const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim();
- const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim();
- const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim();
- if (!accessKeyId || !secretAccessKey || !accountId) {
- onLog(
- `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` +
- `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` +
- `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` +
- `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` +
- `missing files.\n`,
- );
- return 1;
- }
-
- const client = new S3Client({
- region: "auto",
- endpoint: `https://${accountId}.r2.cloudflarestorage.com`,
- credentials: { accessKeyId, secretAccessKey },
- });
-
- onLog(
- `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`,
- );
- try {
- let uploaded = 0;
- let skipped = 0;
- for (const file of files) {
- if (signal.aborted) return 1;
- const key = `${site.siteId}/archives/${file}`;
- const filePath = path.join(stagingDir, file);
- const { size } = await stat(filePath);
- // Skip the re-upload when R2 already holds an object of the same size for
- // this key. The archive cache makes an unchanged channel's zip byte-stable
- // build-to-build, so a same-size object is the same object; a changed
- // channel re-zips to a different size. Avoids re-streaming unchanged
- // multi-MB archives on every deploy. HeadObject is a cheap metadata call.
- try {
- const head = await client.send(
- new HeadObjectCommand({ Bucket: bucket, Key: key }),
- );
- if (head.ContentLength === size) {
- skipped++;
- onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`);
- continue;
- }
- } catch {
- // Not found (or HEAD not permitted) → fall through and upload.
- }
- onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`);
- uploaded++;
- const upload = new Upload({
- client,
- params: {
- Bucket: bucket,
- Key: key,
- Body: createReadStream(filePath),
- ContentType: ARCHIVE_CONTENT_TYPE,
- CacheControl: ARCHIVE_CACHE_CONTROL,
- },
- });
- const onAbort = () => {
- upload.abort().catch(() => {});
- };
- signal.addEventListener("abort", onAbort, { once: true });
- try {
- await upload.done();
- } finally {
- signal.removeEventListener("abort", onAbort);
- }
- onLog(`[archives] ${key} done.\n`);
- }
- onLog(
- `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`,
- );
- return 0;
- } catch (err) {
- onLog(
- `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`,
- );
- return 1;
- } finally {
- client.destroy();
- }
-}
-
-// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare
-// Pages project, streaming into `onLog`, returning the exit code. Runs on the
-// host with the host's Cloudflare credentials (process.env) — deploy never runs
-// inside a container, so container wrangler auth is never needed.
-//
-// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to
-// any branch but the project's production branch as a preview, reachable at the
-// branch alias. Omitting it leaves the argv byte-identical to what production
-// has always run — no `--branch`, so wrangler infers the branch from the
-// checkout, which is the long-standing behaviour (and the long-standing hazard:
-// a "production" deploy run from a non-main checkout silently becomes a
-// preview).
-//
-// Either way, the URL wrangler prints ("Take a peek over at …") earns one
-// terminal line of its own, because the streamed log scrolls and an operator
-// who looked away has nowhere else to find it. A preview also gets the stable
-// branch alias, which is knowable without reading the log at all.
-export async function runDeployIntoLog(
- onLog: (line: string) => void,
- signal: AbortSignal,
- site: Site,
- outDir: string,
- paths: Paths,
- opts?: { previewBranch?: string },
-): Promise<number> {
- const project = site.cloudflareProject as string;
- const previewBranch = opts?.previewBranch?.trim() || undefined;
-
- // Spot the deployment URL as it streams past rather than re-reading the
- // finished log file: the log is the operator's too, and buffering it a second
- // time to grep it would double a big deploy's memory for one line of output.
- let deploymentUrl: string | null = null;
- const watch = (line: string) => {
- if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project);
- onLog(line);
- };
-
- const code = await runChildIntoLog(watch, signal, {
- command: "pnpm",
- args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- EXPORT_PUBLIC_DIR: paths.exportPublicDir,
- SITE_ID: site.siteId,
- },
- });
-
- // Only on success. A URL scraped out of a failed run points at nothing — or
- // worse, at the deployment that is still live.
- if (code === 0) {
- if (previewBranch) {
- const alias = previewAliasUrl(project, previewBranch);
- onLog(
- `[preview] ${alias}` +
- (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") +
- "\n",
- );
- } else if (deploymentUrl) {
- onLog(`[deployed] ${deploymentUrl}\n`);
- }
- }
- return code;
-}
-
-// ---------------------------------------------------------------------------
-// Docker export pipeline (buildPipeline.mode = "docker")
-//
-// Three ordered phases (see DEPLOY_DOCKER.md):
-// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives
-// (warm the shared archive cache). One writer of the shared state.
-// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next
-// build in its own container, read-only over the shared caches, writing only
-// its per-site out/ under exportBuildsDir/<siteId>.
-// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase).
-// ---------------------------------------------------------------------------
-
-// The container engine binary. Defaults to `docker`; podman is a CLI drop-in
-// (and rootless podman yields host-owned outputs without needing `-u`).
-function dockerBin(): string {
- return process.env.DOCKER_BIN?.trim() || "docker";
-}
-
-export type SiteBuildOutcome = { siteId: string; code: number };
-export type SiteDeployOutcome = {
- siteId: string;
- status: "deployed" | "skipped" | "failed";
- reason?: string;
-};
-
-// Cheap probe: is the container engine installed and its daemon reachable? Used
-// to fall back to serial host builds when docker isn't available.
-export async function dockerAvailable(signal: AbortSignal): Promise<boolean> {
- const code = await runChildIntoLog(() => {}, signal, {
- command: dockerBin(),
- args: ["version"],
- cwd: process.cwd(),
- });
- return code === 0;
-}
-
-// Run a host-side export pnpm script (build:data / build:archives) for Phase A.
-// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the
-// shared index/staging land at their canonical export/.export-index location
-// (exactly what the fan-out containers mount read-only).
-async function runHostScript(
- onLog: (line: string) => void,
- signal: AbortSignal,
- paths: Paths,
- script: string,
- extraEnv?: Record<string, string>,
-): Promise<number> {
- return runChildIntoLog(onLog, signal, {
- command: "pnpm",
- args: ["run", script],
- cwd: paths.exportDir,
- env: {
- ...process.env,
- NODE_ENV: "production",
- TRANSCRIPTS_DIR: paths.transcriptsDir,
- ...extraEnv,
- },
- });
-}
-
-// Build (or reuse cached layers of) the per-site build image.
-async function ensureBuildImage(
- onLog: (line: string) => void,
- signal: AbortSignal,
- paths: Paths,
-): Promise<number> {
- const { dockerImage, dockerfile } = getSettings().buildPipeline;
- onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`);
- return runChildIntoLog(onLog, signal, {
- command: dockerBin(),
- args: ["build", "-f", dockerfile, "-t", dockerImage, "."],
- cwd: paths.monorepoRoot,
- });
-}
-
-// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index,
-// and the .export-index staging read-only; mounts the per-site output dir rw.
-// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses
-// next/font/google, which fetches the site's fonts from Google at build time;
-// `--network=none` would fail the build. Isolation still comes from the per-site
-// output dir, the read-only shared mounts, and the non-root `-u` user.
-async function runDockerBuildOne(
- onLog: (line: string) => void,
- signal: AbortSignal,
- siteId: string,
- paths: Paths,
- opts?: { skipArchives?: boolean },
-): Promise<number> {
- const { dockerImage } = getSettings().buildPipeline;
- const siteDir = path.join(paths.exportBuildsDir, siteId);
- // Pre-create the mount target as the host user so container-written files are
- // host-owned (paired with `-u` below), not created root-owned by the daemon.
- await mkdir(siteDir, { recursive: true });
-
- const args = ["run", "--rm", "--init"];
- const uid = typeof process.getuid === "function" ? process.getuid() : null;
- const gid = typeof process.getgid === "function" ? process.getgid() : null;
- if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`);
- // Optional resource caps so N parallel builds (each next build can use ~8 GB)
- // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md.
- const mem = process.env.DOCKER_BUILD_MEMORY?.trim();
- const cpus = process.env.DOCKER_BUILD_CPUS?.trim();
- if (mem) args.push("--memory", mem);
- if (cpus) args.push("--cpus", cpus);
- args.push(
- "-v", `${paths.transcriptsDir}:/data/transcripts:ro`,
- "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`,
- "-v", `${siteDir}:/site`,
- "-e", `SITE_ID=${siteId}`,
- );
- // Mount the host settings.json fresh (build config: archive storage, size caps)
- // rather than relying on a possibly-stale copy — it is NOT baked into the image.
- if (existsSync(paths.settingsFile)) {
- args.push(
- "-v", `${paths.settingsFile}:/data/settings.json:ro`,
- "-e", "SETTINGS_FILE=/data/settings.json",
- );
- }
- // Mount the entrypoint fresh over the baked copy so a script tweak takes effect
- // without an image rebuild (the image still bakes it as a fallback).
- const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh");
- if (existsSync(entrypoint)) {
- args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`);
- }
- if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0");
- args.push(dockerImage);
-
- return runChildIntoLog(onLog, signal, {
- command: dockerBin(),
- args,
- cwd: paths.monorepoRoot,
- label: `[${siteId}] `,
- });
-}
-
-// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an
-// infrastructure failure (data phase / archive warm / image build) that aborts
-// the whole run before any site could build; per-site build failures are
-// returned, not thrown, so one bad site never blocks the rest.
-export async function runDockerBuildAllPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- sites: Site[],
- paths: Paths,
- opts?: { skipArchives?: boolean },
-): Promise<SiteBuildOutcome[]> {
- const { maxParallelBuilds } = getSettings().buildPipeline;
-
- // --- Phase A: shared data + archive cache (host, serial) ---
- onLog("=== Phase A: shared data + archive cache (host) ===");
- const dataCode = await runHostScript(onLog, signal, paths, "build:data");
- if (signal.aborted) return [];
- if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`);
- if (!opts?.skipArchives) {
- const archCode = await runHostScript(onLog, signal, paths, "build:archives");
- if (signal.aborted) return [];
- if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`);
- }
-
- const imgCode = await ensureBuildImage(onLog, signal, paths);
- if (signal.aborted) return [];
- if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`);
-
- // --- Phase B: per-site fan-out (containers, parallel) ---
- onLog(
- `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`,
- );
- return runWithConcurrency(sites, maxParallelBuilds, async (site) => {
- if (signal.aborted) return { siteId: site.siteId, code: 1 };
- const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts);
- onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`);
- return { siteId: site.siteId, code };
- });
-}
-
-// Phase C: deploy each built site SERIALLY on the host, after the build barrier.
-// Partial-failure tolerant — a site that fails to upload/deploy is recorded and
-// the loop continues. Sites that failed to build, or have no Cloudflare project,
-// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site;
-// basic fallback: export/out).
-export async function runDockerDeployAllPhase(
- onLog: (line: string) => void,
- signal: AbortSignal,
- sites: Site[],
- builtOk: Set<string>,
- paths: Paths,
- outDirFor: (siteId: string) => string,
-): Promise<SiteDeployOutcome[]> {
- const outcomes: SiteDeployOutcome[] = [];
- for (const site of sites) {
- if (signal.aborted) break;
- if (!builtOk.has(site.siteId)) {
- onLog(`[${site.siteId}] deploy skipped — build failed`);
- outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" });
- continue;
- }
- if (!site.cloudflareProject) {
- onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`);
- outcomes.push({
- siteId: site.siteId,
- status: "skipped",
- reason: "no cloudflareProject",
- });
- continue;
- }
- onLog(`=== Deploy ${site.siteId} ===`);
- const uploadCode = await runArchiveUploadIntoLog(
- onLog,
- signal,
- site,
- paths,
- dockerSiteStagingDir(paths, site.siteId),
- );
- if (signal.aborted) break;
- if (uploadCode !== 0) {
- onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`);
- outcomes.push({
- siteId: site.siteId,
- status: "failed",
- reason: `R2 upload exit ${uploadCode}`,
- });
- continue;
- }
- const deployCode = await runDeployIntoLog(
- onLog,
- signal,
- site,
- outDirFor(site.siteId),
- paths,
- );
- if (signal.aborted) break;
- if (deployCode !== 0) {
- onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`);
- outcomes.push({
- siteId: site.siteId,
- status: "failed",
- reason: `deploy exit ${deployCode}`,
- });
- continue;
- }
- onLog(`[${site.siteId}] deployed.`);
- outcomes.push({ siteId: site.siteId, status: "deployed" });
- }
- return outcomes;
-}
-
-// Bounded-concurrency map over a fixed work set, preserving input order in the
-// results. No external dep; a fresh worker pulls the next index until exhausted.
-async function runWithConcurrency<T, R>(
- items: T[],
- limit: number,
- worker: (item: T) => Promise<R>,
-): Promise<R[]> {
- const results: R[] = new Array(items.length);
- let next = 0;
- const width = Math.max(1, Math.min(limit, items.length));
- const runners = Array.from({ length: width }, async () => {
- while (true) {
- const i = next++;
- if (i >= items.length) break;
- results[i] = await worker(items[i]);
- }
- });
- await Promise.all(runners);
- return results;
-}
diff --git a/editor/app/sites/lib/deployAction.ts b/editor/app/sites/lib/deployAction.ts
@@ -13,7 +13,7 @@ import {
resolveOutDir,
runArchiveUploadIntoLog,
runDeployIntoLog,
-} from "./buildDeployCore";
+} from "yt-dlp-transcript-common/publish/build";
const DEPLOY_QUEUE = "deploy";
diff --git a/editor/package.json b/editor/package.json
@@ -14,8 +14,6 @@
"e2e:ui": "playwright test --ui"
},
"dependencies": {
- "@aws-sdk/client-s3": "^3.1080.0",
- "@aws-sdk/lib-storage": "^3.1080.0",
"@sindresorhus/slugify": "^3.0.0",
"lucide-react": "^1.16.0",
"markdown-to-jsx": "^7.7.4",
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -5318,7 +5318,7 @@ file). On exit 0 the job log gets ONE line: `[preview] <alias> (this deployment:
namespace and the keys are `<siteId>/archives/<file>.zip` either way. Cheap in
practice (the upload skips any object R2 already holds at the same size), but a
*changed* archive replaces the one production's manifest links to.
-`PREVIEW_SHARES_ARCHIVES_NOTICE` (`editor/app/sites/lib/buildDeployCore.ts`) is
+`PREVIEW_SHARES_ARCHIVES_NOTICE` (`common/publish/build.ts`, moved from the editor in one-core Phase 4 slice 1) is
logged once at the top of every preview deploy, by both actions.
**Three surfaces, one rule.** `deployExportAction(siteId, { previewBranch })` and
diff --git a/plans/release-6.md b/plans/release-6.md
@@ -45,3 +45,64 @@ queue. Numbers: **none**. No file format changed; `availability.json` keeps its
is the same for routes this commit did not touch. The state-tree encoding may have drifted with
Next 16.2; this commit only replaced the paths, so it was not investigated.
- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as in release 5.
+
+### Phase 4 slice 1, as shipped — `buildDeployCore` to the core as `common/publish/build.ts` (2026-09-25)
+
+Branch `one-core/phase-4-s1` off `main` `93dcb532`; `main` moved to `4d97049f` (the follow-ups
+slice) mid-slice and was merged before the final gates. `editor/app/sites/lib/buildDeployCore.ts`
+(578 lines) imported nothing from `editor/**`, so it moved **unchanged** to
+`common/publish/build.ts`, the path `one-core.md` Phase 4 item 1 names. Only its header comment
+and its three `yt-dlp-transcript-common/*` imports (now relative, `../jobs`, `../lib`, as the rest
+of common writes them) changed. Every export, log line, exit code and path is the same. No helper
+had to come down from the editor with it.
+
+| sha | what |
+|---|---|
+| `9cb35c37` | `git mv` to `common/publish/build.ts`. `sites/lib/buildAction.ts` and `deployAction.ts` import `yt-dlp-transcript-common/publish/build`. **No re-export is left at the old path**: no spec, script or tool imports it (`git grep buildDeployCore` finds only docs and plans). `@aws-sdk/client-s3` and `@aws-sdk/lib-storage` (`^3.1080.0`) move from `editor/package.json` to common's `dependencies`. As in the follow-ups slice, a plain `pnpm install` re-resolved unrelated peer suffixes (`supports-color`), so the lockfile change was applied by hand (the two importer entries move from `editor:` to `common:`, 12 lines) and verified with `pnpm install --frozen-lockfile`. `common/package.json` also gains `"./publish/*": "./publish/*.ts"` in `exports` (the editor resolves common through `exports`, so the new directory needs its own pattern) and `publish` in the `test` glob. `architecture.test.ts` learns the layer: `publish/` may not import `views/` or `components/`, and `lib/` and `components/` may not import `publish/`. No back-edge was found and `ALLOWED` did not grow. New `publish/build.test.ts` (3) pins `resolveOutDir`, `dockerSiteOutDir` and `dockerSiteStagingDir`. `DEPLOY_CLOUDFLARE.md`'s pointer for `ARCHIVE_CACHE_CONTROL` (it still named the pre-IA `editor/app/deploy/` path) names the new home |
+| `2a31a863` | merge `main` `4d97049f` (release 6 follow-ups). Clean: that slice's `exports` lines sit between `./components/*` and `./lib/*`, and this one sits after `./views/*`. Then `pnpm install --frozen-lockfile` to create the new umtool → common link (`--offline` failed for want of cached metadata for `@next/env`; the online frozen install changed nothing on disk in git) |
+| `c8af2165` | this record, the `[Unreleased]` bullet, `plans/FACTS.md`'s `PREVIEW_SHARES_ARCHIVES_NOTICE` path |
+| `f2507e89` | (review fix) `architecture.test.ts`: the `jobs` and `controller` rows forbid `publish/` too, since dispatch sits below publish, and the failure message says so. Nothing imports that way, so it stays green |
+| *(this commit)* | (review fix) this record: how the editor build loads the SDK, `Dockerfile.build`'s extra install, the commit table |
+
+**Gates** on the merged tree (`2a31a863`, worktree root). After the review fix: tsc clean,
+common **1754/1754**, editor unit **72/72**. No e2e was rerun, since the fix changes no behaviour. tsc
+(`pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit`) clean, and clean before
+`9cb35c37` too. common **1754/1754**: 1750 + 1 (follow-ups) + 3 (`publish/build.test.ts`); it was
+1753 before the merge. Editor unit **72/72**. test:scripts **159 pass + 1 skip** (the follow-ups
+count). mcp **219/219**. `pnpm --filter editor exec next build` ok (compiled in 18.6 s). With
+editor no longer depending on the SDK, `@aws-sdk/lib-storage` is bundled, and `@aws-sdk/client-s3`
+stays external: it loads through the symlink Next creates at
+`editor/.next/node_modules/@aws-sdk/client-s3-<hash>`, which points into `common/node_modules`.
+`pnpm --filter export exec next build` ok (7.8 s). The worktree's `export/public/archives` link
+was dangling (the primary has no `archives/` at the moment), so it was removed before the builds.
+EDITOR e2e `build deploy-page site-publish-preview sites-crud cut-release channel-build-toggle`
+(all six exist; `$T/p4-specs.txt`): **29 passed, 0 failed, 1.8 min**, with no wait in the queue.
+Numbers: `plans/tools/phase3-files-numbers.ts` over one frozen copy of the corpus
+(`FREEZE_TO`, 71 configs, 1,763 sidecars), `main` `93dcb532` against this branch (both
+before and after the merge): **diff empty** (3,858 lines each). `pnpm ops build-site` /
+`build-deploy` / `deploy-site` post to `editor/app/api/ops/*`, which call the same actions. They
+compile in the editor build and were not run, because a real build against the corpus is
+forbidden.
+
+**Left from Phase 4 item 1, by name.** The brief scoped this slice to the move. What `one-core.md`
+item 1 also lists is still to do, and most of it needs item 2's CLI first:
+- the named entry points `buildSite(id, opts)`, `deploySite`, `buildAll(mode)`, `composeHub` and
+ `composeHomepage`. The orchestration that would become them (`runManagedFunction` jobs, the
+ docker fallback, per-site queues) still lives in the editor's `"use server"`
+ `sites/lib/buildAction.ts` / `deployAction.ts`;
+- `docker/build-site.sh` and `publish-site.sh` calling the CLI;
+- export's `build` / `build:nodata` twins becoming one script with `--nodata`;
+- `build:hub` getting a CLI and an editor entry.
+
+**Found and left.**
+- **`common/package.json` `exports` was on the list of files owned by the follow-ups slice.** The
+ one-line `./publish/*` addition cannot be avoided if the spec's `common/publish/` path is kept:
+ the editor resolves common through `exports`. It was made anyway and is its own line. `git
+ merge-tree` against `one-core/r6-followups` was clean before that slice landed, and the real
+ merge was clean too.
+- **`Dockerfile.build`'s export image now installs the AWS SDK too**, because it installs
+ common's dependencies. That adds weight to the image and breaks nothing; the export never
+ imports `publish/`.
+- `plans/STATE.md` still says Phase 4 is next and that `buildDeployCore.ts` imports nothing from
+ the editor (`:73`, `:1485`). Status is the parent's to write, so those lines were not edited.
+- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as in release 5.
diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml
@@ -14,6 +14,12 @@ importers:
common:
dependencies:
+ '@aws-sdk/client-s3':
+ specifier: ^3.1080.0
+ version: 3.1080.0
+ '@aws-sdk/lib-storage':
+ specifier: ^3.1080.0
+ version: 3.1080.0(@aws-sdk/client-s3@3.1080.0)
'@sindresorhus/slugify':
specifier: ^3.0.0
version: 3.0.0
@@ -114,12 +120,6 @@ importers:
editor:
dependencies:
- '@aws-sdk/client-s3':
- specifier: ^3.1080.0
- version: 3.1080.0
- '@aws-sdk/lib-storage':
- specifier: ^3.1080.0
- version: 3.1080.0(@aws-sdk/client-s3@3.1080.0)
'@sindresorhus/slugify':
specifier: ^3.0.0
version: 3.0.0