/// // Cloudflare Worker: public read proxy for the oversize archive zips stored in R2. // // ONE Worker serves EVERY export site. The R2 key is namespaced by site id // (`/archives/.zip`), so this is a pure passthrough that maps the // request path straight to the bucket key — there is nothing per-site about it. // Deploy it ONCE to a free `.workers.dev` subdomain (no custom domain, no // domain purchase, no WHOIS), then point every site's "Archive overflow public // URL" at that single subdomain. See ../PUBLISH.md. // // Why a Worker instead of the raw r2.dev URL: it gives us tunable, in-code rate // limiting (the cost/abuse backstop — Cloudflare's dashboard rate-limit rules // require a zone/custom domain) plus our own edge caching, all on the free tier. interface RateLimiter { limit(options: { key: string }): Promise<{ success: boolean }>; } export interface Env { BUCKET: R2Bucket; RATE_LIMITER: RateLimiter; } // Only ever serve archive zips: "/archives/.zip". Anything else // 404s, so the proxy can't be turned into a general read oracle over the bucket. const KEY_RE = /^[A-Za-z0-9._-]+\/archives\/[A-Za-z0-9._-]+\.zip$/; export default { async fetch( request: Request, env: Env, ctx: ExecutionContext, ): Promise { if (request.method !== "GET" && request.method !== "HEAD") { return new Response("Method not allowed", { status: 405, headers: { allow: "GET, HEAD" }, }); } const url = new URL(request.url); const key = decodeURIComponent(url.pathname.replace(/^\/+/, "")); if (!KEY_RE.test(key)) { return new Response("Not found", { status: 404 }); } // Per-IP + per-file rate limit (per edge location) — the abuse/cost backstop. const ip = request.headers.get("cf-connecting-ip") ?? "anon"; const { success } = await env.RATE_LIMITER.limit({ key: `${ip}:${key}` }); if (!success) { return new Response("Too many requests", { status: 429, headers: { "retry-after": "60" }, }); } const isRange = request.headers.has("range"); const cache = caches.default; const cacheKey = new Request(url.toString(), { method: "GET" }); // Full (non-range) requests can be served from — and stored in — the edge // cache, so repeat downloads skip R2 entirely (no billable Class B op). if (!isRange) { const cached = await cache.match(cacheKey); if (cached) { return request.method === "HEAD" ? new Response(null, { status: cached.status, headers: cached.headers }) : cached; } } const object = await env.BUCKET.get(key, { range: request.headers, onlyIf: request.headers, }); if (object === null) { return new Response("Not found", { status: 404 }); } const headers = new Headers(); object.writeHttpMetadata(headers); headers.set("etag", object.httpEtag); headers.set("accept-ranges", "bytes"); if (!headers.has("cache-control")) { headers.set("cache-control", "public, max-age=3600"); } if (!headers.has("content-type")) { headers.set("content-type", "application/zip"); } // No body ⇒ an onlyIf precondition matched (e.g. If-None-Match) ⇒ 304. const body = "body" in object ? (object as R2ObjectBody).body : null; if (!body) { return new Response(null, { status: 304, headers }); } let status = 200; const range = (object as R2ObjectBody).range as | { offset?: number; length?: number } | undefined; if (isRange && range && typeof range.offset === "number") { const offset = range.offset; const length = typeof range.length === "number" ? range.length : object.size - offset; headers.set("content-range", `bytes ${offset}-${offset + length - 1}/${object.size}`); headers.set("content-length", String(length)); status = 206; } else { headers.set("content-length", String(object.size)); } const response = new Response(request.method === "HEAD" ? null : body, { status, headers, }); // Cache full 200 GET responses at the edge (best-effort — Cloudflare skips // ones that are too large or otherwise non-cacheable). Never cache 206. if (status === 200 && request.method === "GET") { ctx.waitUntil(cache.put(cacheKey, response.clone()).catch(() => {})); } return response; }, };