///
// Cloudflare Worker: public read proxy for the oversize archive zips stored in R2.
//
// ONE Worker serves EVERY export site. The R2 key is namespaced by site id
// (`/archives/.zip`), so this is a pure passthrough that maps the
// request path straight to the bucket key — there is nothing per-site about it.
// Deploy it ONCE to a free `.workers.dev` subdomain (no custom domain, no
// domain purchase, no WHOIS), then point every site's "Archive overflow public
// URL" at that single subdomain. See ../PUBLISH.md.
//
// Why a Worker instead of the raw r2.dev URL: it gives us tunable, in-code rate
// limiting (the cost/abuse backstop — Cloudflare's dashboard rate-limit rules
// require a zone/custom domain) plus our own edge caching, all on the free tier.
interface RateLimiter {
limit(options: { key: string }): Promise<{ success: boolean }>;
}
export interface Env {
BUCKET: R2Bucket;
RATE_LIMITER: RateLimiter;
}
// Only ever serve archive zips: "/archives/.zip". Anything else
// 404s, so the proxy can't be turned into a general read oracle over the bucket.
const KEY_RE = /^[A-Za-z0-9._-]+\/archives\/[A-Za-z0-9._-]+\.zip$/;
export default {
async fetch(
request: Request,
env: Env,
ctx: ExecutionContext,
): Promise {
if (request.method !== "GET" && request.method !== "HEAD") {
return new Response("Method not allowed", {
status: 405,
headers: { allow: "GET, HEAD" },
});
}
const url = new URL(request.url);
const key = decodeURIComponent(url.pathname.replace(/^\/+/, ""));
if (!KEY_RE.test(key)) {
return new Response("Not found", { status: 404 });
}
// Per-IP + per-file rate limit (per edge location) — the abuse/cost backstop.
const ip = request.headers.get("cf-connecting-ip") ?? "anon";
const { success } = await env.RATE_LIMITER.limit({ key: `${ip}:${key}` });
if (!success) {
return new Response("Too many requests", {
status: 429,
headers: { "retry-after": "60" },
});
}
const isRange = request.headers.has("range");
const cache = caches.default;
const cacheKey = new Request(url.toString(), { method: "GET" });
// Full (non-range) requests can be served from — and stored in — the edge
// cache, so repeat downloads skip R2 entirely (no billable Class B op).
if (!isRange) {
const cached = await cache.match(cacheKey);
if (cached) {
return request.method === "HEAD"
? new Response(null, { status: cached.status, headers: cached.headers })
: cached;
}
}
const object = await env.BUCKET.get(key, {
range: request.headers,
onlyIf: request.headers,
});
if (object === null) {
return new Response("Not found", { status: 404 });
}
const headers = new Headers();
object.writeHttpMetadata(headers);
headers.set("etag", object.httpEtag);
headers.set("accept-ranges", "bytes");
if (!headers.has("cache-control")) {
headers.set("cache-control", "public, max-age=3600");
}
if (!headers.has("content-type")) {
headers.set("content-type", "application/zip");
}
// No body ⇒ an onlyIf precondition matched (e.g. If-None-Match) ⇒ 304.
const body = "body" in object ? (object as R2ObjectBody).body : null;
if (!body) {
return new Response(null, { status: 304, headers });
}
let status = 200;
const range = (object as R2ObjectBody).range as
| { offset?: number; length?: number }
| undefined;
if (isRange && range && typeof range.offset === "number") {
const offset = range.offset;
const length =
typeof range.length === "number" ? range.length : object.size - offset;
headers.set("content-range", `bytes ${offset}-${offset + length - 1}/${object.size}`);
headers.set("content-length", String(length));
status = 206;
} else {
headers.set("content-length", String(object.size));
}
const response = new Response(request.method === "HEAD" ? null : body, {
status,
headers,
});
// Cache full 200 GET responses at the edge (best-effort — Cloudflare skips
// ones that are too large or otherwise non-cacheable). Never cache 206.
if (status === 200 && request.method === "GET") {
ctx.waitUntil(cache.put(cacheKey, response.clone()).catch(() => {}));
}
return response;
},
};