Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 10b4f62c5918e7a114fc0df16a58483da3a54e47
parent 05877ffb6d9756dbbb8173225125dd05bf403794
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 25 Sep 2026 00:45:27 -0400

common: buildDeployCore moves to the core as publish/build.ts

`editor/app/sites/lib/buildDeployCore.ts` imported nothing from the editor,
so it moves unchanged to `common/publish/build.ts` (one-core Phase 4 item 1).
The two callers, `sites/lib/buildAction.ts` and `deployAction.ts`, import
`yt-dlp-transcript-common/publish/build`; no re-export stays behind, since no
spec imports the old path. Log lines, exit codes and paths are identical.

`@aws-sdk/client-s3` and `@aws-sdk/lib-storage` move from editor's
dependencies to common's; the lockfile change is those two importer entries
and nothing else (`pnpm install --frozen-lockfile` passes).

common/package.json gains `./publish/*` in `exports` (the editor resolves
through it) and `publish` in the test glob. architecture.test.ts learns the
layer: `publish/` may not import `views/` or `components/`, and `lib/` and
`components/` may not import `publish/`. A unit test pins the three pure
path helpers.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
MDEPLOY_CLOUDFLARE.md | 2+-
Mcommon/architecture.test.ts | 14++++++++++----
Mcommon/package.json | 5++++-
Acommon/publish/build.test.ts | 43+++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/build.ts | 582++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/sites/lib/buildAction.ts | 2+-
Deditor/app/sites/lib/buildDeployCore.ts | 578------------------------------------------------------------------------------
Meditor/app/sites/lib/deployAction.ts | 2+-
Meditor/package.json | 2--
Mpnpm-lock.yaml | 12++++++------
10 files changed, 648 insertions(+), 594 deletions(-)

diff --git a/DEPLOY_CLOUDFLARE.md b/DEPLOY_CLOUDFLARE.md @@ -242,7 +242,7 @@ small and commented if you want to adjust them. The archive `Cache-Control` is also set at upload time (`Cache-Control: public, max-age=3600`, constant `ARCHIVE_CACHE_CONTROL` in -`editor/app/deploy/buildDeployCore.ts`); the Worker reads it back when caching. Archive +`common/publish/build.ts`); the Worker reads it back when caching. Archive filenames are stable and overwritten in place on re-deploy, so this 1-hour bound is what keeps a re-uploaded archive from being served stale for long — raise it if your archives rarely change. diff --git a/common/architecture.test.ts b/common/architecture.test.ts @@ -35,15 +35,20 @@ const FORBIDDEN: Record<string, readonly string[]> = { // searchQuery for a mode union. All four inverted — the pipeline and the two // types moved down, and the fetch and the memo are now injected — so the // guard turns on with nothing added to the allow-list to pay for it. - lib: ["controller", "jobs", "components", "views"], + lib: ["controller", "jobs", "components", "views", "publish"], jobs: ["controller", "views"], controller: ["views"], - components: ["controller", "jobs", "ytdlp"], + components: ["controller", "jobs", "ytdlp", "publish"], // `views/` is the view-model layer (one-core phase 3 slice 1): pure functions // that fold live state into a payload. It sits ABOVE dispatch and BELOW ui, // so it may read lib/, controller/ and jobs/ for pure helpers and types, and // may not reach up into components/ or sideways into the entry points. views: ["components", "ytdlp", "bin", "social"], + // `publish/` is the publish layer (one-core phase 4 slice 1): building a site, + // uploading its archives, deploying it. It sits above dispatch and below views, + // so it may read lib/, jobs/ and controller/, and may not reach up into + // views/ or components/. + publish: ["views", "components"], }; // The directories walked. `bin/` and `social/` are scanned so a back-edge cannot @@ -57,6 +62,7 @@ const ROOTS = [ "ytdlp", "bin", "views", + "publish", ]; // Today's back-edges, `<file> -> <imported module>`, each with why it is still @@ -164,10 +170,10 @@ test("no new back-edges between common's layers", async () => { assert.deepEqual( unexpected, [], - `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, ` + + `NEW back-edge(s) in common/. lib/ may not import controller/, jobs/, publish/, ` + `components/ or views/; ` + `jobs/ and controller/ may not import views/; components/ may not ` + - `import controller/, jobs/ or ytdlp/; views/ may not import ` + + `import controller/, jobs/, ytdlp/ or publish/; publish/ may not import views/ or components/; views/ may not import ` + `components/. Move the type or the function down a ` + `layer instead of adding it to ALLOWED. Found: ${unexpected.join(", ")}`, ); diff --git a/common/package.json b/common/package.json @@ -35,6 +35,7 @@ "./controller/*": "./controller/*.ts", "./jobs/*": "./jobs/*.ts", "./views/*": "./views/*.ts", + "./publish/*": "./publish/*.ts", "./social/*": "./social/*.ts", "./ytdlp/*": "./ytdlp/*.ts", "./bin/*": "./bin/*.ts", @@ -42,9 +43,11 @@ "./styles/*": "./styles/*.ts" }, "scripts": { - "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views}/*/*.test.ts\"" + "test": "tsx --test \"*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*.test.ts\" \"{lib,controller,jobs,social,ytdlp,components,views,publish}/*/*.test.ts\"" }, "dependencies": { + "@aws-sdk/client-s3": "^3.1080.0", + "@aws-sdk/lib-storage": "^3.1080.0", "@sindresorhus/slugify": "^3.0.0", "@tanstack/react-query": "^5.99.1", "@tanstack/react-virtual": "^3.13.0", diff --git a/common/publish/build.test.ts b/common/publish/build.test.ts @@ -0,0 +1,43 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import type { Paths } from "../lib/paths"; +import { + dockerSiteOutDir, + dockerSiteStagingDir, + resolveOutDir, +} from "./build"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common test +// +// The three pure path helpers of the publish layer. Their outputs are the +// on-disk contract the build and deploy phases share (and that +// `docker/build-site.sh` mounts), so they are pinned here exactly. + +const paths = { + exportDir: "/repo/export", + exportBuildsDir: "/repo/export/.export-builds", +} as Paths; + +test("resolveOutDir is export/out whatever the site", () => { + assert.equal(resolveOutDir("jeralyzer", paths), "/repo/export/out"); + assert.equal(resolveOutDir("anilyzer", paths), "/repo/export/out"); +}); + +test("dockerSiteOutDir is a per-site out/ under exportBuildsDir", () => { + assert.equal( + dockerSiteOutDir(paths, "jeralyzer"), + "/repo/export/.export-builds/jeralyzer/out", + ); + assert.notEqual( + dockerSiteOutDir(paths, "jeralyzer"), + dockerSiteOutDir(paths, "anilyzer"), + ); +}); + +test("dockerSiteStagingDir nests .r2-staging/<site>/archives under the site's build dir", () => { + assert.equal( + dockerSiteStagingDir(paths, "jeralyzer"), + "/repo/export/.export-builds/jeralyzer/.r2-staging/jeralyzer/archives", + ); +}); diff --git a/common/publish/build.ts b/common/publish/build.ts @@ -0,0 +1,582 @@ +// The publish layer's build/deploy primitives: one site's host build, the +// docker per-site fan-out, the R2 archive upload and the Pages deploy. Moved +// here from the editor (`editor/app/sites/lib/buildDeployCore.ts`) in one-core +// Phase 4 slice 1, unchanged; the editor's build/deploy server actions +// (`editor/app/sites/lib/{buildAction,deployAction}.ts`) call them as jobs. +// +// These take an `onLog` callback and an AbortSignal, so they CANNOT live in a +// "use server" module (every export there becomes a server action, which +// forbids non-serializable args). Keep them as plain helpers. + +import path from "node:path"; +import { mkdir, readdir, stat } from "node:fs/promises"; +import { createReadStream, existsSync } from "node:fs"; +import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3"; +import { Upload } from "@aws-sdk/lib-storage"; +import { runChildIntoLog } from "../jobs/runChild"; +import { + deploymentUrlIn, + pagesDeployArgs, + previewAliasUrl, +} from "../lib/pagesDeploy"; +import type { Paths } from "../lib/paths"; +import { getSettings } from "../lib/settings"; +import type { Site } from "../lib/site"; + +// Where the basic (host) build writes the static bundle to deploy: the fixed +// export/out, composed one site at a time. The docker fan-out writes per-site +// out/ dirs instead — see dockerSiteOutDir. +export function resolveOutDir(_siteId: string, paths: Paths): string { + return path.join(paths.exportDir, "out"); +} + +// Where a docker per-site container writes its built bundle (mounted as /site/out +// inside the container). Isolated per site so parallel builds never collide. +export function dockerSiteOutDir(paths: Paths, siteId: string): string { + return path.join(paths.exportBuildsDir, siteId, "out"); +} + +// Where a docker per-site container stages oversize archives for R2 upload. The +// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/ +// public), so compose writes .r2-staging as its sibling under exportBuildsDir. +export function dockerSiteStagingDir(paths: Paths, siteId: string): string { + return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives"); +} + +// Run the basic (host) build phase for one site, streaming into `onLog`, +// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream +// on the build queue, since the export/ tree is shared). This is the single-site +// build for basic mode, and the fallback the all-sites docker action drops to +// when no container engine is available. The parallel per-site container path +// lives in runDockerBuildAllPhase. +export async function runBuildPhase( + onLog: (line: string) => void, + signal: AbortSignal, + siteId: string, + paths: Paths, + opts?: { skipData?: boolean; skipArchives?: boolean }, +): Promise<number> { + // When skipping the data rebuild, run the `build:nodata` script instead of + // `build`. `build:nodata` has the same body (compose:site + next build) but a + // different name, so npm's `prebuild` lifecycle hook (which runs build:data = + // index/stats/templates) does NOT fire — we compose from the existing + // .export-index staging. Assumes a prior full build produced that staging. + const skipData = opts?.skipData === true; + if (skipData) { + onLog( + "[notice] Skipping data rebuild (index/stats/charts) — composing from " + + "existing .export-index staging.\n", + ); + } + // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes + // compose-site skip generation this build regardless of the global/site flags. + const skipArchives = opts?.skipArchives === true; + if (skipArchives) { + onLog("[notice] Skipping archive-zip generation for this build.\n"); + } + return runChildIntoLog(onLog, signal, { + command: "pnpm", + args: ["run", skipData ? "build:nodata" : "build"], + cwd: paths.exportDir, + env: { + ...process.env, + NODE_ENV: "production", + TRANSCRIPTS_DIR: paths.transcriptsDir, + EXPORT_PUBLIC_DIR: paths.exportPublicDir, + SITE_ID: siteId, + ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}), + }, + }); +} + +// Where the basic (host) compose staged this site's oversize archives for R2 +// upload. Kept outside export/public so they never ship as Pages assets. Mirrors +// the path composeArchives writes to in common/bin/compose-site.ts. The docker +// fan-out uses dockerSiteStagingDir instead, passed explicitly. +function archiveStagingDir(siteId: string, paths: Paths): string { + return path.join( + path.dirname(paths.exportPublicDir), + ".r2-staging", + siteId, + "archives", + ); +} + +// Said once at the top of every preview deploy, because the one thing a preview +// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a +// preview's oversize archives overwrite the keys production's manifest points +// at. That is cheap and harmless in practice — the upload skips any object R2 +// already holds at the same size, and an unchanged channel re-zips byte-stable +// — but "harmless because of a size check" is exactly the kind of thing an +// operator should be told rather than left to discover. +export const PREVIEW_SHARES_ARCHIVES_NOTICE = + "[notice] A preview shares the production R2 archive bucket — unchanged " + + "archives are skipped, so this is cheap, but a CHANGED archive replaces the " + + "one production links to.\n"; + +// Cache-Control set on every uploaded archive. Served through a Cloudflare custom +// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead +// of hitting R2 (each origin GET is a billable Class B op), which is the main cost +// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood +// absorption against re-deployed archives (stable filenames, overwritten in place) +// going stale; raise it if your archives rarely change. +const ARCHIVE_CACHE_CONTROL = "public, max-age=3600"; + +// MIME type stored on each uploaded archive. Archives are `.zip` bundles. +const ARCHIVE_CONTENT_TYPE = "application/zip"; + +// Upload this site's staged oversize archives to the configured R2 bucket, so the +// remote URLs the served manifest points at actually resolve. Uploads go through +// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) — +// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real +// live-chat archives blow past, whereas multipart streams any size. Keys match +// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must +// run BEFORE the Pages deploy so the manifest never points at a missing object. +// +// No-op (returns 0) when overflow storage isn't configured or nothing was staged. +// Credentials come from the host env (NOT the settings JSON — secrets don't +// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID +// (for the S3 endpoint). If a bucket is configured and files are staged but the +// credentials are missing, we fail (return non-zero) so the deploy aborts rather +// than shipping a manifest that points at un-uploaded objects. +export async function runArchiveUploadIntoLog( + onLog: (line: string) => void, + signal: AbortSignal, + site: Site, + paths: Paths, + // Where the oversize archives were staged. Defaults to the basic host location; + // the docker deploy phase passes dockerSiteStagingDir since its container wrote + // .r2-staging under exportBuildsDir/<siteId> instead. + stagingDirOverride?: string, +): Promise<number> { + const bucket = getSettings().archiveStorage?.bucket?.trim(); + if (!bucket) return 0; + + const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths); + let files: string[]; + try { + files = (await readdir(stagingDir)).filter((f) => !f.startsWith(".")); + } catch { + // No staging dir → nothing oversize this build. + return 0; + } + if (files.length === 0) return 0; + + const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim(); + const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim(); + const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim(); + if (!accessKeyId || !secretAccessKey || !accountId) { + onLog( + `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` + + `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` + + `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` + + `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` + + `missing files.\n`, + ); + return 1; + } + + const client = new S3Client({ + region: "auto", + endpoint: `https://${accountId}.r2.cloudflarestorage.com`, + credentials: { accessKeyId, secretAccessKey }, + }); + + onLog( + `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`, + ); + try { + let uploaded = 0; + let skipped = 0; + for (const file of files) { + if (signal.aborted) return 1; + const key = `${site.siteId}/archives/${file}`; + const filePath = path.join(stagingDir, file); + const { size } = await stat(filePath); + // Skip the re-upload when R2 already holds an object of the same size for + // this key. The archive cache makes an unchanged channel's zip byte-stable + // build-to-build, so a same-size object is the same object; a changed + // channel re-zips to a different size. Avoids re-streaming unchanged + // multi-MB archives on every deploy. HeadObject is a cheap metadata call. + try { + const head = await client.send( + new HeadObjectCommand({ Bucket: bucket, Key: key }), + ); + if (head.ContentLength === size) { + skipped++; + onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`); + continue; + } + } catch { + // Not found (or HEAD not permitted) → fall through and upload. + } + onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`); + uploaded++; + const upload = new Upload({ + client, + params: { + Bucket: bucket, + Key: key, + Body: createReadStream(filePath), + ContentType: ARCHIVE_CONTENT_TYPE, + CacheControl: ARCHIVE_CACHE_CONTROL, + }, + }); + const onAbort = () => { + upload.abort().catch(() => {}); + }; + signal.addEventListener("abort", onAbort, { once: true }); + try { + await upload.done(); + } finally { + signal.removeEventListener("abort", onAbort); + } + onLog(`[archives] ${key} done.\n`); + } + onLog( + `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`, + ); + return 0; + } catch (err) { + onLog( + `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`, + ); + return 1; + } finally { + client.destroy(); + } +} + +// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare +// Pages project, streaming into `onLog`, returning the exit code. Runs on the +// host with the host's Cloudflare credentials (process.env) — deploy never runs +// inside a container, so container wrangler auth is never needed. +// +// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to +// any branch but the project's production branch as a preview, reachable at the +// branch alias. Omitting it leaves the argv byte-identical to what production +// has always run — no `--branch`, so wrangler infers the branch from the +// checkout, which is the long-standing behaviour (and the long-standing hazard: +// a "production" deploy run from a non-main checkout silently becomes a +// preview). +// +// Either way, the URL wrangler prints ("Take a peek over at …") earns one +// terminal line of its own, because the streamed log scrolls and an operator +// who looked away has nowhere else to find it. A preview also gets the stable +// branch alias, which is knowable without reading the log at all. +export async function runDeployIntoLog( + onLog: (line: string) => void, + signal: AbortSignal, + site: Site, + outDir: string, + paths: Paths, + opts?: { previewBranch?: string }, +): Promise<number> { + const project = site.cloudflareProject as string; + const previewBranch = opts?.previewBranch?.trim() || undefined; + + // Spot the deployment URL as it streams past rather than re-reading the + // finished log file: the log is the operator's too, and buffering it a second + // time to grep it would double a big deploy's memory for one line of output. + let deploymentUrl: string | null = null; + const watch = (line: string) => { + if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project); + onLog(line); + }; + + const code = await runChildIntoLog(watch, signal, { + command: "pnpm", + args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })], + cwd: paths.exportDir, + env: { + ...process.env, + NODE_ENV: "production", + TRANSCRIPTS_DIR: paths.transcriptsDir, + EXPORT_PUBLIC_DIR: paths.exportPublicDir, + SITE_ID: site.siteId, + }, + }); + + // Only on success. A URL scraped out of a failed run points at nothing — or + // worse, at the deployment that is still live. + if (code === 0) { + if (previewBranch) { + const alias = previewAliasUrl(project, previewBranch); + onLog( + `[preview] ${alias}` + + (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") + + "\n", + ); + } else if (deploymentUrl) { + onLog(`[deployed] ${deploymentUrl}\n`); + } + } + return code; +} + +// --------------------------------------------------------------------------- +// Docker export pipeline (buildPipeline.mode = "docker") +// +// Three ordered phases (see DEPLOY_DOCKER.md): +// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives +// (warm the shared archive cache). One writer of the shared state. +// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next +// build in its own container, read-only over the shared caches, writing only +// its per-site out/ under exportBuildsDir/<siteId>. +// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase). +// --------------------------------------------------------------------------- + +// The container engine binary. Defaults to `docker`; podman is a CLI drop-in +// (and rootless podman yields host-owned outputs without needing `-u`). +function dockerBin(): string { + return process.env.DOCKER_BIN?.trim() || "docker"; +} + +export type SiteBuildOutcome = { siteId: string; code: number }; +export type SiteDeployOutcome = { + siteId: string; + status: "deployed" | "skipped" | "failed"; + reason?: string; +}; + +// Cheap probe: is the container engine installed and its daemon reachable? Used +// to fall back to serial host builds when docker isn't available. +export async function dockerAvailable(signal: AbortSignal): Promise<boolean> { + const code = await runChildIntoLog(() => {}, signal, { + command: dockerBin(), + args: ["version"], + cwd: process.cwd(), + }); + return code === 0; +} + +// Run a host-side export pnpm script (build:data / build:archives) for Phase A. +// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the +// shared index/staging land at their canonical export/.export-index location +// (exactly what the fan-out containers mount read-only). +async function runHostScript( + onLog: (line: string) => void, + signal: AbortSignal, + paths: Paths, + script: string, + extraEnv?: Record<string, string>, +): Promise<number> { + return runChildIntoLog(onLog, signal, { + command: "pnpm", + args: ["run", script], + cwd: paths.exportDir, + env: { + ...process.env, + NODE_ENV: "production", + TRANSCRIPTS_DIR: paths.transcriptsDir, + ...extraEnv, + }, + }); +} + +// Build (or reuse cached layers of) the per-site build image. +async function ensureBuildImage( + onLog: (line: string) => void, + signal: AbortSignal, + paths: Paths, +): Promise<number> { + const { dockerImage, dockerfile } = getSettings().buildPipeline; + onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`); + return runChildIntoLog(onLog, signal, { + command: dockerBin(), + args: ["build", "-f", dockerfile, "-t", dockerImage, "."], + cwd: paths.monorepoRoot, + }); +} + +// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index, +// and the .export-index staging read-only; mounts the per-site output dir rw. +// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses +// next/font/google, which fetches the site's fonts from Google at build time; +// `--network=none` would fail the build. Isolation still comes from the per-site +// output dir, the read-only shared mounts, and the non-root `-u` user. +async function runDockerBuildOne( + onLog: (line: string) => void, + signal: AbortSignal, + siteId: string, + paths: Paths, + opts?: { skipArchives?: boolean }, +): Promise<number> { + const { dockerImage } = getSettings().buildPipeline; + const siteDir = path.join(paths.exportBuildsDir, siteId); + // Pre-create the mount target as the host user so container-written files are + // host-owned (paired with `-u` below), not created root-owned by the daemon. + await mkdir(siteDir, { recursive: true }); + + const args = ["run", "--rm", "--init"]; + const uid = typeof process.getuid === "function" ? process.getuid() : null; + const gid = typeof process.getgid === "function" ? process.getgid() : null; + if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`); + // Optional resource caps so N parallel builds (each next build can use ~8 GB) + // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md. + const mem = process.env.DOCKER_BUILD_MEMORY?.trim(); + const cpus = process.env.DOCKER_BUILD_CPUS?.trim(); + if (mem) args.push("--memory", mem); + if (cpus) args.push("--cpus", cpus); + args.push( + "-v", `${paths.transcriptsDir}:/data/transcripts:ro`, + "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`, + "-v", `${siteDir}:/site`, + "-e", `SITE_ID=${siteId}`, + ); + // Mount the host settings.json fresh (build config: archive storage, size caps) + // rather than relying on a possibly-stale copy — it is NOT baked into the image. + if (existsSync(paths.settingsFile)) { + args.push( + "-v", `${paths.settingsFile}:/data/settings.json:ro`, + "-e", "SETTINGS_FILE=/data/settings.json", + ); + } + // Mount the entrypoint fresh over the baked copy so a script tweak takes effect + // without an image rebuild (the image still bakes it as a fallback). + const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh"); + if (existsSync(entrypoint)) { + args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`); + } + if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0"); + args.push(dockerImage); + + return runChildIntoLog(onLog, signal, { + command: dockerBin(), + args, + cwd: paths.monorepoRoot, + label: `[${siteId}] `, + }); +} + +// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an +// infrastructure failure (data phase / archive warm / image build) that aborts +// the whole run before any site could build; per-site build failures are +// returned, not thrown, so one bad site never blocks the rest. +export async function runDockerBuildAllPhase( + onLog: (line: string) => void, + signal: AbortSignal, + sites: Site[], + paths: Paths, + opts?: { skipArchives?: boolean }, +): Promise<SiteBuildOutcome[]> { + const { maxParallelBuilds } = getSettings().buildPipeline; + + // --- Phase A: shared data + archive cache (host, serial) --- + onLog("=== Phase A: shared data + archive cache (host) ==="); + const dataCode = await runHostScript(onLog, signal, paths, "build:data"); + if (signal.aborted) return []; + if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`); + if (!opts?.skipArchives) { + const archCode = await runHostScript(onLog, signal, paths, "build:archives"); + if (signal.aborted) return []; + if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`); + } + + const imgCode = await ensureBuildImage(onLog, signal, paths); + if (signal.aborted) return []; + if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`); + + // --- Phase B: per-site fan-out (containers, parallel) --- + onLog( + `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`, + ); + return runWithConcurrency(sites, maxParallelBuilds, async (site) => { + if (signal.aborted) return { siteId: site.siteId, code: 1 }; + const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts); + onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`); + return { siteId: site.siteId, code }; + }); +} + +// Phase C: deploy each built site SERIALLY on the host, after the build barrier. +// Partial-failure tolerant — a site that fails to upload/deploy is recorded and +// the loop continues. Sites that failed to build, or have no Cloudflare project, +// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site; +// basic fallback: export/out). +export async function runDockerDeployAllPhase( + onLog: (line: string) => void, + signal: AbortSignal, + sites: Site[], + builtOk: Set<string>, + paths: Paths, + outDirFor: (siteId: string) => string, +): Promise<SiteDeployOutcome[]> { + const outcomes: SiteDeployOutcome[] = []; + for (const site of sites) { + if (signal.aborted) break; + if (!builtOk.has(site.siteId)) { + onLog(`[${site.siteId}] deploy skipped — build failed`); + outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" }); + continue; + } + if (!site.cloudflareProject) { + onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`); + outcomes.push({ + siteId: site.siteId, + status: "skipped", + reason: "no cloudflareProject", + }); + continue; + } + onLog(`=== Deploy ${site.siteId} ===`); + const uploadCode = await runArchiveUploadIntoLog( + onLog, + signal, + site, + paths, + dockerSiteStagingDir(paths, site.siteId), + ); + if (signal.aborted) break; + if (uploadCode !== 0) { + onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`); + outcomes.push({ + siteId: site.siteId, + status: "failed", + reason: `R2 upload exit ${uploadCode}`, + }); + continue; + } + const deployCode = await runDeployIntoLog( + onLog, + signal, + site, + outDirFor(site.siteId), + paths, + ); + if (signal.aborted) break; + if (deployCode !== 0) { + onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`); + outcomes.push({ + siteId: site.siteId, + status: "failed", + reason: `deploy exit ${deployCode}`, + }); + continue; + } + onLog(`[${site.siteId}] deployed.`); + outcomes.push({ siteId: site.siteId, status: "deployed" }); + } + return outcomes; +} + +// Bounded-concurrency map over a fixed work set, preserving input order in the +// results. No external dep; a fresh worker pulls the next index until exhausted. +async function runWithConcurrency<T, R>( + items: T[], + limit: number, + worker: (item: T) => Promise<R>, +): Promise<R[]> { + const results: R[] = new Array(items.length); + let next = 0; + const width = Math.max(1, Math.min(limit, items.length)); + const runners = Array.from({ length: width }, async () => { + while (true) { + const i = next++; + if (i >= items.length) break; + results[i] = await worker(items[i]); + } + }); + await Promise.all(runners); + return results; +} diff --git a/editor/app/sites/lib/buildAction.ts b/editor/app/sites/lib/buildAction.ts @@ -33,7 +33,7 @@ import { runDockerDeployAllPhase, type SiteBuildOutcome, type SiteDeployOutcome, -} from "./buildDeployCore"; +} from "yt-dlp-transcript-common/publish/build"; const DEFAULT_BUILD_QUEUE = "build"; const DEPLOY_QUEUE = "deploy"; diff --git a/editor/app/sites/lib/buildDeployCore.ts b/editor/app/sites/lib/buildDeployCore.ts @@ -1,578 +0,0 @@ -// Shared, server-only build/deploy primitives used by the build/deploy server -// actions (buildAction.ts, deployAction.ts). These take an `onLog` callback and -// an AbortSignal, so they CANNOT live in a "use server" module (every export -// there becomes a server action, which forbids non-serializable args). Keep them -// here as plain helpers and import them into the thin action wrappers. - -import path from "node:path"; -import { mkdir, readdir, stat } from "node:fs/promises"; -import { createReadStream, existsSync } from "node:fs"; -import { S3Client, HeadObjectCommand } from "@aws-sdk/client-s3"; -import { Upload } from "@aws-sdk/lib-storage"; -import { runChildIntoLog } from "yt-dlp-transcript-common/jobs/runChild"; -import { - deploymentUrlIn, - pagesDeployArgs, - previewAliasUrl, -} from "yt-dlp-transcript-common/lib/pagesDeploy"; -import type { Paths } from "yt-dlp-transcript-common/lib/paths"; -import { getSettings } from "yt-dlp-transcript-common/lib/settings"; -import type { Site } from "yt-dlp-transcript-common/lib/site"; - -// Where the basic (host) build writes the static bundle to deploy: the fixed -// export/out, composed one site at a time. The docker fan-out writes per-site -// out/ dirs instead — see dockerSiteOutDir. -export function resolveOutDir(_siteId: string, paths: Paths): string { - return path.join(paths.exportDir, "out"); -} - -// Where a docker per-site container writes its built bundle (mounted as /site/out -// inside the container). Isolated per site so parallel builds never collide. -export function dockerSiteOutDir(paths: Paths, siteId: string): string { - return path.join(paths.exportBuildsDir, siteId, "out"); -} - -// Where a docker per-site container stages oversize archives for R2 upload. The -// container's EXPORT_PUBLIC_DIR is /site/public (host exportBuildsDir/<siteId>/ -// public), so compose writes .r2-staging as its sibling under exportBuildsDir. -export function dockerSiteStagingDir(paths: Paths, siteId: string): string { - return path.join(paths.exportBuildsDir, siteId, ".r2-staging", siteId, "archives"); -} - -// Run the basic (host) build phase for one site, streaming into `onLog`, -// returning the exit code. Runs `pnpm run build` in export/ (serialized upstream -// on the build queue, since the export/ tree is shared). This is the single-site -// build for basic mode, and the fallback the all-sites docker action drops to -// when no container engine is available. The parallel per-site container path -// lives in runDockerBuildAllPhase. -export async function runBuildPhase( - onLog: (line: string) => void, - signal: AbortSignal, - siteId: string, - paths: Paths, - opts?: { skipData?: boolean; skipArchives?: boolean }, -): Promise<number> { - // When skipping the data rebuild, run the `build:nodata` script instead of - // `build`. `build:nodata` has the same body (compose:site + next build) but a - // different name, so npm's `prebuild` lifecycle hook (which runs build:data = - // index/stats/templates) does NOT fire — we compose from the existing - // .export-index staging. Assumes a prior full build produced that staging. - const skipData = opts?.skipData === true; - if (skipData) { - onLog( - "[notice] Skipping data rebuild (index/stats/charts) — composing from " + - "existing .export-index staging.\n", - ); - } - // Per-build opt-out for the bulk-download archive zips. BUILD_ARCHIVES=0 makes - // compose-site skip generation this build regardless of the global/site flags. - const skipArchives = opts?.skipArchives === true; - if (skipArchives) { - onLog("[notice] Skipping archive-zip generation for this build.\n"); - } - return runChildIntoLog(onLog, signal, { - command: "pnpm", - args: ["run", skipData ? "build:nodata" : "build"], - cwd: paths.exportDir, - env: { - ...process.env, - NODE_ENV: "production", - TRANSCRIPTS_DIR: paths.transcriptsDir, - EXPORT_PUBLIC_DIR: paths.exportPublicDir, - SITE_ID: siteId, - ...(skipArchives ? { BUILD_ARCHIVES: "0" } : {}), - }, - }); -} - -// Where the basic (host) compose staged this site's oversize archives for R2 -// upload. Kept outside export/public so they never ship as Pages assets. Mirrors -// the path composeArchives writes to in common/bin/compose-site.ts. The docker -// fan-out uses dockerSiteStagingDir instead, passed explicitly. -function archiveStagingDir(siteId: string, paths: Paths): string { - return path.join( - path.dirname(paths.exportPublicDir), - ".r2-staging", - siteId, - "archives", - ); -} - -// Said once at the top of every preview deploy, because the one thing a preview -// does NOT isolate is the archive bucket: R2 has no per-branch namespace, so a -// preview's oversize archives overwrite the keys production's manifest points -// at. That is cheap and harmless in practice — the upload skips any object R2 -// already holds at the same size, and an unchanged channel re-zips byte-stable -// — but "harmless because of a size check" is exactly the kind of thing an -// operator should be told rather than left to discover. -export const PREVIEW_SHARES_ARCHIVES_NOTICE = - "[notice] A preview shares the production R2 archive bucket — unchanged " + - "archives are skipped, so this is cheap, but a CHANGED archive replaces the " + - "one production links to.\n"; - -// Cache-Control set on every uploaded archive. Served through a Cloudflare custom -// domain, this lets the CDN absorb repeated/abusive downloads at the edge instead -// of hitting R2 (each origin GET is a billable Class B op), which is the main cost -// defense for public archives — see DEPLOY_CLOUDFLARE.md. 1h balances flood -// absorption against re-deployed archives (stable filenames, overwritten in place) -// going stale; raise it if your archives rarely change. -const ARCHIVE_CACHE_CONTROL = "public, max-age=3600"; - -// MIME type stored on each uploaded archive. Archives are `.zip` bundles. -const ARCHIVE_CONTENT_TYPE = "application/zip"; - -// Upload this site's staged oversize archives to the configured R2 bucket, so the -// remote URLs the served manifest points at actually resolve. Uploads go through -// R2's S3 API with the AWS SDK's multipart uploader (`@aws-sdk/lib-storage`) — -// `wrangler r2 object put` caps single-file uploads at 300 MiB, which real -// live-chat archives blow past, whereas multipart streams any size. Keys match -// what composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must -// run BEFORE the Pages deploy so the manifest never points at a missing object. -// -// No-op (returns 0) when overflow storage isn't configured or nothing was staged. -// Credentials come from the host env (NOT the settings JSON — secrets don't -// belong there): R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID -// (for the S3 endpoint). If a bucket is configured and files are staged but the -// credentials are missing, we fail (return non-zero) so the deploy aborts rather -// than shipping a manifest that points at un-uploaded objects. -export async function runArchiveUploadIntoLog( - onLog: (line: string) => void, - signal: AbortSignal, - site: Site, - paths: Paths, - // Where the oversize archives were staged. Defaults to the basic host location; - // the docker deploy phase passes dockerSiteStagingDir since its container wrote - // .r2-staging under exportBuildsDir/<siteId> instead. - stagingDirOverride?: string, -): Promise<number> { - const bucket = getSettings().archiveStorage?.bucket?.trim(); - if (!bucket) return 0; - - const stagingDir = stagingDirOverride ?? archiveStagingDir(site.siteId, paths); - let files: string[]; - try { - files = (await readdir(stagingDir)).filter((f) => !f.startsWith(".")); - } catch { - // No staging dir → nothing oversize this build. - return 0; - } - if (files.length === 0) return 0; - - const accessKeyId = process.env.R2_ACCESS_KEY_ID?.trim(); - const secretAccessKey = process.env.R2_SECRET_ACCESS_KEY?.trim(); - const accountId = process.env.CLOUDFLARE_ACCOUNT_ID?.trim(); - if (!accessKeyId || !secretAccessKey || !accountId) { - onLog( - `[archives] ${files.length} oversize archive(s) need uploading to R2 bucket ` + - `"${bucket}", but R2 S3 credentials are missing. Set R2_ACCESS_KEY_ID, ` + - `R2_SECRET_ACCESS_KEY, and CLOUDFLARE_ACCOUNT_ID in the environment — see ` + - `DEPLOY_CLOUDFLARE.md. Aborting before deploy so the site never links to ` + - `missing files.\n`, - ); - return 1; - } - - const client = new S3Client({ - region: "auto", - endpoint: `https://${accountId}.r2.cloudflarestorage.com`, - credentials: { accessKeyId, secretAccessKey }, - }); - - onLog( - `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}" via the S3 API…\n`, - ); - try { - let uploaded = 0; - let skipped = 0; - for (const file of files) { - if (signal.aborted) return 1; - const key = `${site.siteId}/archives/${file}`; - const filePath = path.join(stagingDir, file); - const { size } = await stat(filePath); - // Skip the re-upload when R2 already holds an object of the same size for - // this key. The archive cache makes an unchanged channel's zip byte-stable - // build-to-build, so a same-size object is the same object; a changed - // channel re-zips to a different size. Avoids re-streaming unchanged - // multi-MB archives on every deploy. HeadObject is a cheap metadata call. - try { - const head = await client.send( - new HeadObjectCommand({ Bucket: bucket, Key: key }), - ); - if (head.ContentLength === size) { - skipped++; - onLog(`[archives] ${key} unchanged — already in R2, skipping.\n`); - continue; - } - } catch { - // Not found (or HEAD not permitted) → fall through and upload. - } - onLog(`[archives] ${key} (${(size / 1e6).toFixed(1)} MB)…\n`); - uploaded++; - const upload = new Upload({ - client, - params: { - Bucket: bucket, - Key: key, - Body: createReadStream(filePath), - ContentType: ARCHIVE_CONTENT_TYPE, - CacheControl: ARCHIVE_CACHE_CONTROL, - }, - }); - const onAbort = () => { - upload.abort().catch(() => {}); - }; - signal.addEventListener("abort", onAbort, { once: true }); - try { - await upload.done(); - } finally { - signal.removeEventListener("abort", onAbort); - } - onLog(`[archives] ${key} done.\n`); - } - onLog( - `[archives] R2 upload complete (${uploaded} uploaded, ${skipped} unchanged).\n`, - ); - return 0; - } catch (err) { - onLog( - `[archives] R2 upload failed: ${err instanceof Error ? err.message : String(err)}\n`, - ); - return 1; - } finally { - client.destroy(); - } -} - -// Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare -// Pages project, streaming into `onLog`, returning the exit code. Runs on the -// host with the host's Cloudflare credentials (process.env) — deploy never runs -// inside a container, so container wrangler auth is never needed. -// -// `opts.previewBranch` makes it a PREVIEW deploy: Cloudflare treats a deploy to -// any branch but the project's production branch as a preview, reachable at the -// branch alias. Omitting it leaves the argv byte-identical to what production -// has always run — no `--branch`, so wrangler infers the branch from the -// checkout, which is the long-standing behaviour (and the long-standing hazard: -// a "production" deploy run from a non-main checkout silently becomes a -// preview). -// -// Either way, the URL wrangler prints ("Take a peek over at …") earns one -// terminal line of its own, because the streamed log scrolls and an operator -// who looked away has nowhere else to find it. A preview also gets the stable -// branch alias, which is knowable without reading the log at all. -export async function runDeployIntoLog( - onLog: (line: string) => void, - signal: AbortSignal, - site: Site, - outDir: string, - paths: Paths, - opts?: { previewBranch?: string }, -): Promise<number> { - const project = site.cloudflareProject as string; - const previewBranch = opts?.previewBranch?.trim() || undefined; - - // Spot the deployment URL as it streams past rather than re-reading the - // finished log file: the log is the operator's too, and buffering it a second - // time to grep it would double a big deploy's memory for one line of output. - let deploymentUrl: string | null = null; - const watch = (line: string) => { - if (deploymentUrl === null) deploymentUrl = deploymentUrlIn(line, project); - onLog(line); - }; - - const code = await runChildIntoLog(watch, signal, { - command: "pnpm", - args: ["dlx", ...pagesDeployArgs({ outDir, project, previewBranch })], - cwd: paths.exportDir, - env: { - ...process.env, - NODE_ENV: "production", - TRANSCRIPTS_DIR: paths.transcriptsDir, - EXPORT_PUBLIC_DIR: paths.exportPublicDir, - SITE_ID: site.siteId, - }, - }); - - // Only on success. A URL scraped out of a failed run points at nothing — or - // worse, at the deployment that is still live. - if (code === 0) { - if (previewBranch) { - const alias = previewAliasUrl(project, previewBranch); - onLog( - `[preview] ${alias}` + - (deploymentUrl ? ` (this deployment: ${deploymentUrl})` : "") + - "\n", - ); - } else if (deploymentUrl) { - onLog(`[deployed] ${deploymentUrl}\n`); - } - } - return code; -} - -// --------------------------------------------------------------------------- -// Docker export pipeline (buildPipeline.mode = "docker") -// -// Three ordered phases (see DEPLOY_DOCKER.md): -// A) HOST, serial: build:data (shared LMDB + .export-index) then build:archives -// (warm the shared archive cache). One writer of the shared state. -// B) CONTAINERS, parallel (cap maxParallelBuilds): each site's compose + next -// build in its own container, read-only over the shared caches, writing only -// its per-site out/ under exportBuildsDir/<siteId>. -// C) HOST, serial: deploy each built site (handled by runDockerDeployAllPhase). -// --------------------------------------------------------------------------- - -// The container engine binary. Defaults to `docker`; podman is a CLI drop-in -// (and rootless podman yields host-owned outputs without needing `-u`). -function dockerBin(): string { - return process.env.DOCKER_BIN?.trim() || "docker"; -} - -export type SiteBuildOutcome = { siteId: string; code: number }; -export type SiteDeployOutcome = { - siteId: string; - status: "deployed" | "skipped" | "failed"; - reason?: string; -}; - -// Cheap probe: is the container engine installed and its daemon reachable? Used -// to fall back to serial host builds when docker isn't available. -export async function dockerAvailable(signal: AbortSignal): Promise<boolean> { - const code = await runChildIntoLog(() => {}, signal, { - command: dockerBin(), - args: ["version"], - cwd: process.cwd(), - }); - return code === 0; -} - -// Run a host-side export pnpm script (build:data / build:archives) for Phase A. -// These are pool-wide: no SITE_ID, and no EXPORT_PUBLIC_DIR override so the -// shared index/staging land at their canonical export/.export-index location -// (exactly what the fan-out containers mount read-only). -async function runHostScript( - onLog: (line: string) => void, - signal: AbortSignal, - paths: Paths, - script: string, - extraEnv?: Record<string, string>, -): Promise<number> { - return runChildIntoLog(onLog, signal, { - command: "pnpm", - args: ["run", script], - cwd: paths.exportDir, - env: { - ...process.env, - NODE_ENV: "production", - TRANSCRIPTS_DIR: paths.transcriptsDir, - ...extraEnv, - }, - }); -} - -// Build (or reuse cached layers of) the per-site build image. -async function ensureBuildImage( - onLog: (line: string) => void, - signal: AbortSignal, - paths: Paths, -): Promise<number> { - const { dockerImage, dockerfile } = getSettings().buildPipeline; - onLog(`[docker] building image "${dockerImage}" from ${dockerfile} (cached layers reused)`); - return runChildIntoLog(onLog, signal, { - command: dockerBin(), - args: ["build", "-f", dockerfile, "-t", dockerImage, "."], - cwd: paths.monorepoRoot, - }); -} - -// Run ONE site's build in a container. Mounts the corpus, the shared LMDB index, -// and the .export-index staging read-only; mounts the per-site output dir rw. -// Streams with a [siteId] prefix. Network is left ENABLED — `next build` uses -// next/font/google, which fetches the site's fonts from Google at build time; -// `--network=none` would fail the build. Isolation still comes from the per-site -// output dir, the read-only shared mounts, and the non-root `-u` user. -async function runDockerBuildOne( - onLog: (line: string) => void, - signal: AbortSignal, - siteId: string, - paths: Paths, - opts?: { skipArchives?: boolean }, -): Promise<number> { - const { dockerImage } = getSettings().buildPipeline; - const siteDir = path.join(paths.exportBuildsDir, siteId); - // Pre-create the mount target as the host user so container-written files are - // host-owned (paired with `-u` below), not created root-owned by the daemon. - await mkdir(siteDir, { recursive: true }); - - const args = ["run", "--rm", "--init"]; - const uid = typeof process.getuid === "function" ? process.getuid() : null; - const gid = typeof process.getgid === "function" ? process.getgid() : null; - if (uid !== null && gid !== null) args.push("-u", `${uid}:${gid}`); - // Optional resource caps so N parallel builds (each next build can use ~8 GB) - // don't OOM the host. Sized by the operator; see DEPLOY_DOCKER.md. - const mem = process.env.DOCKER_BUILD_MEMORY?.trim(); - const cpus = process.env.DOCKER_BUILD_CPUS?.trim(); - if (mem) args.push("--memory", mem); - if (cpus) args.push("--cpus", cpus); - args.push( - "-v", `${paths.transcriptsDir}:/data/transcripts:ro`, - "-v", `${paths.exportIndexDir}:/data/export/.export-index:ro`, - "-v", `${siteDir}:/site`, - "-e", `SITE_ID=${siteId}`, - ); - // Mount the host settings.json fresh (build config: archive storage, size caps) - // rather than relying on a possibly-stale copy — it is NOT baked into the image. - if (existsSync(paths.settingsFile)) { - args.push( - "-v", `${paths.settingsFile}:/data/settings.json:ro`, - "-e", "SETTINGS_FILE=/data/settings.json", - ); - } - // Mount the entrypoint fresh over the baked copy so a script tweak takes effect - // without an image rebuild (the image still bakes it as a fallback). - const entrypoint = path.join(paths.monorepoRoot, "docker", "build-site.sh"); - if (existsSync(entrypoint)) { - args.push("-v", `${entrypoint}:/repo/docker/build-site.sh:ro`); - } - if (opts?.skipArchives) args.push("-e", "BUILD_ARCHIVES=0"); - args.push(dockerImage); - - return runChildIntoLog(onLog, signal, { - command: dockerBin(), - args, - cwd: paths.monorepoRoot, - label: `[${siteId}] `, - }); -} - -// Phases A + B. Returns each site's build exit code (0 = ok). Throws only on an -// infrastructure failure (data phase / archive warm / image build) that aborts -// the whole run before any site could build; per-site build failures are -// returned, not thrown, so one bad site never blocks the rest. -export async function runDockerBuildAllPhase( - onLog: (line: string) => void, - signal: AbortSignal, - sites: Site[], - paths: Paths, - opts?: { skipArchives?: boolean }, -): Promise<SiteBuildOutcome[]> { - const { maxParallelBuilds } = getSettings().buildPipeline; - - // --- Phase A: shared data + archive cache (host, serial) --- - onLog("=== Phase A: shared data + archive cache (host) ==="); - const dataCode = await runHostScript(onLog, signal, paths, "build:data"); - if (signal.aborted) return []; - if (dataCode !== 0) throw new Error(`Data phase failed (exit ${dataCode}).`); - if (!opts?.skipArchives) { - const archCode = await runHostScript(onLog, signal, paths, "build:archives"); - if (signal.aborted) return []; - if (archCode !== 0) throw new Error(`Archive cache warm failed (exit ${archCode}).`); - } - - const imgCode = await ensureBuildImage(onLog, signal, paths); - if (signal.aborted) return []; - if (imgCode !== 0) throw new Error(`Docker image build failed (exit ${imgCode}).`); - - // --- Phase B: per-site fan-out (containers, parallel) --- - onLog( - `=== Phase B: building ${sites.length} site(s), up to ${maxParallelBuilds} in parallel ===`, - ); - return runWithConcurrency(sites, maxParallelBuilds, async (site) => { - if (signal.aborted) return { siteId: site.siteId, code: 1 }; - const code = await runDockerBuildOne(onLog, signal, site.siteId, paths, opts); - onLog(`[${site.siteId}] build ${code === 0 ? "ok" : `FAILED (exit ${code})`}`); - return { siteId: site.siteId, code }; - }); -} - -// Phase C: deploy each built site SERIALLY on the host, after the build barrier. -// Partial-failure tolerant — a site that fails to upload/deploy is recorded and -// the loop continues. Sites that failed to build, or have no Cloudflare project, -// are skipped. `outDirFor` resolves each site's built bundle (docker: per-site; -// basic fallback: export/out). -export async function runDockerDeployAllPhase( - onLog: (line: string) => void, - signal: AbortSignal, - sites: Site[], - builtOk: Set<string>, - paths: Paths, - outDirFor: (siteId: string) => string, -): Promise<SiteDeployOutcome[]> { - const outcomes: SiteDeployOutcome[] = []; - for (const site of sites) { - if (signal.aborted) break; - if (!builtOk.has(site.siteId)) { - onLog(`[${site.siteId}] deploy skipped — build failed`); - outcomes.push({ siteId: site.siteId, status: "skipped", reason: "build failed" }); - continue; - } - if (!site.cloudflareProject) { - onLog(`[${site.siteId}] deploy skipped — no Cloudflare project configured`); - outcomes.push({ - siteId: site.siteId, - status: "skipped", - reason: "no cloudflareProject", - }); - continue; - } - onLog(`=== Deploy ${site.siteId} ===`); - const uploadCode = await runArchiveUploadIntoLog( - onLog, - signal, - site, - paths, - dockerSiteStagingDir(paths, site.siteId), - ); - if (signal.aborted) break; - if (uploadCode !== 0) { - onLog(`[${site.siteId}] deploy FAILED — R2 upload exit ${uploadCode}`); - outcomes.push({ - siteId: site.siteId, - status: "failed", - reason: `R2 upload exit ${uploadCode}`, - }); - continue; - } - const deployCode = await runDeployIntoLog( - onLog, - signal, - site, - outDirFor(site.siteId), - paths, - ); - if (signal.aborted) break; - if (deployCode !== 0) { - onLog(`[${site.siteId}] deploy FAILED — exit ${deployCode}`); - outcomes.push({ - siteId: site.siteId, - status: "failed", - reason: `deploy exit ${deployCode}`, - }); - continue; - } - onLog(`[${site.siteId}] deployed.`); - outcomes.push({ siteId: site.siteId, status: "deployed" }); - } - return outcomes; -} - -// Bounded-concurrency map over a fixed work set, preserving input order in the -// results. No external dep; a fresh worker pulls the next index until exhausted. -async function runWithConcurrency<T, R>( - items: T[], - limit: number, - worker: (item: T) => Promise<R>, -): Promise<R[]> { - const results: R[] = new Array(items.length); - let next = 0; - const width = Math.max(1, Math.min(limit, items.length)); - const runners = Array.from({ length: width }, async () => { - while (true) { - const i = next++; - if (i >= items.length) break; - results[i] = await worker(items[i]); - } - }); - await Promise.all(runners); - return results; -} diff --git a/editor/app/sites/lib/deployAction.ts b/editor/app/sites/lib/deployAction.ts @@ -13,7 +13,7 @@ import { resolveOutDir, runArchiveUploadIntoLog, runDeployIntoLog, -} from "./buildDeployCore"; +} from "yt-dlp-transcript-common/publish/build"; const DEPLOY_QUEUE = "deploy"; diff --git a/editor/package.json b/editor/package.json @@ -14,8 +14,6 @@ "e2e:ui": "playwright test --ui" }, "dependencies": { - "@aws-sdk/client-s3": "^3.1080.0", - "@aws-sdk/lib-storage": "^3.1080.0", "@sindresorhus/slugify": "^3.0.0", "lucide-react": "^1.16.0", "markdown-to-jsx": "^7.7.4", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml @@ -14,6 +14,12 @@ importers: common: dependencies: + '@aws-sdk/client-s3': + specifier: ^3.1080.0 + version: 3.1080.0 + '@aws-sdk/lib-storage': + specifier: ^3.1080.0 + version: 3.1080.0(@aws-sdk/client-s3@3.1080.0) '@sindresorhus/slugify': specifier: ^3.0.0 version: 3.0.0 @@ -114,12 +120,6 @@ importers: editor: dependencies: - '@aws-sdk/client-s3': - specifier: ^3.1080.0 - version: 3.1080.0 - '@aws-sdk/lib-storage': - specifier: ^3.1080.0 - version: 3.1080.0(@aws-sdk/client-s3@3.1080.0) '@sindresorhus/slugify': specifier: ^3.0.0 version: 3.0.0