// A ZIP READ IN PLACE (release 21 D1, controller/attachMedia.ts). // // The archives media is attached from are tens of GB and mostly STORED (no // compression: video does not deflate), so the useful operation is "copy this // entry's bytes out", and for a stored entry that is a byte range of the zip // file itself — no temp dir, no extraction tool, no second copy. A deflated // entry is inflated on the way out. Anything else (bzip2, LZMA, encryption) is // refused by name rather than half-read. // // Only the CENTRAL DIRECTORY is trusted for sizes and offsets (the local // header's sizes are zero when bit 3 is set), with ZIP64 throughout: a 36 GB // archive's later offsets do not fit 32 bits. The local header is read only // for its own name/extra lengths, which is where the data starts. // // READ-ONLY. Every open here is `r`; nothing writes to the archive. // // SERVER-ONLY (node:fs). import { createReadStream, createWriteStream } from "node:fs"; import { open, type FileHandle } from "node:fs/promises"; import { createHash } from "node:crypto"; import { Transform, type TransformCallback } from "node:stream"; import { pipeline } from "node:stream/promises"; import { createInflateRaw, crc32, inflateRawSync } from "node:zlib"; import { Readable } from "node:stream"; export type ZipEntry = { // The entry's path inside the archive, "/"-separated, as stored. name: string; isDirectory: boolean; // 0 = stored, 8 = deflated; anything else is refused at extraction. method: number; compressedSize: number; size: number; crc32: number; localHeaderOffset: number; encrypted: boolean; }; const EOCD_SIG = 0x06054b50; const ZIP64_LOCATOR_SIG = 0x07064b50; const ZIP64_EOCD_SIG = 0x06064b50; const CDH_SIG = 0x02014b50; const LFH_SIG = 0x04034b50; const U32_MAX = 0xffffffff; const U16_MAX = 0xffff; export class ZipFormatError extends Error {} async function readAt(fh: FileHandle, offset: number, length: number): Promise { const buf = Buffer.alloc(length); let got = 0; while (got < length) { const { bytesRead } = await fh.read(buf, got, length - got, offset + got); if (bytesRead === 0) break; got += bytesRead; } return got === length ? buf : buf.subarray(0, got); } // A 64-bit little-endian value as a JS number. Every offset and size in a // real archive is far below 2^53; one that is not is a corrupt file. function u64(buf: Buffer, at: number): number { const v = buf.readBigUInt64LE(at); if (v > BigInt(Number.MAX_SAFE_INTEGER)) { throw new ZipFormatError("a ZIP64 value exceeds 2^53"); } return Number(v); } const utf8Strict = new TextDecoder("utf-8", { fatal: true }); // Bit 11 says UTF-8. Without it the spec says CP437, but in practice archives // written on Linux and by most tools are UTF-8 without the flag; valid UTF-8 // is read as UTF-8, anything else as latin1 (identical to CP437 for ASCII). function decodeName(raw: Buffer, utf8Flag: boolean): string { if (utf8Flag) return raw.toString("utf8"); try { return utf8Strict.decode(raw); } catch { return raw.toString("latin1"); } } type Extra = { id: number; data: Buffer }; function parseExtras(buf: Buffer): Extra[] { const out: Extra[] = []; let at = 0; while (at + 4 <= buf.length) { const id = buf.readUInt16LE(at); const len = buf.readUInt16LE(at + 2); const data = buf.subarray(at + 4, at + 4 + len); out.push({ id, data }); at += 4 + len; } return out; } async function findEocd(fh: FileHandle, fileSize: number): Promise<{ buf: Buffer; at: number }> { // EOCD is 22 bytes plus a comment of up to 65535. const tail = Math.min(fileSize, 22 + 0xffff); const buf = await readAt(fh, fileSize - tail, tail); for (let i = buf.length - 22; i >= 0; i--) { if (buf.readUInt32LE(i) === EOCD_SIG) { return { buf: buf.subarray(i), at: fileSize - tail + i }; } } throw new ZipFormatError("not a zip file (no end-of-central-directory record)"); } // Every entry of the archive, from its central directory. export async function readZipEntries(file: string): Promise { const fh = await open(file, "r"); try { const { size: fileSize } = await fh.stat(); const eocd = await findEocd(fh, fileSize); let count = eocd.buf.readUInt16LE(10); let cdSize = eocd.buf.readUInt32LE(12); let cdOffset = eocd.buf.readUInt32LE(16); if (count === U16_MAX || cdSize === U32_MAX || cdOffset === U32_MAX) { if (eocd.at < 20) throw new ZipFormatError("ZIP64 locator missing"); const loc = await readAt(fh, eocd.at - 20, 20); if (loc.readUInt32LE(0) !== ZIP64_LOCATOR_SIG) { throw new ZipFormatError("ZIP64 locator missing"); } const z64At = u64(loc, 8); const z64 = await readAt(fh, z64At, 56); if (z64.readUInt32LE(0) !== ZIP64_EOCD_SIG) { throw new ZipFormatError("ZIP64 end-of-central-directory record missing"); } count = u64(z64, 32); cdSize = u64(z64, 40); cdOffset = u64(z64, 48); } const cd = await readAt(fh, cdOffset, cdSize); const entries: ZipEntry[] = []; let at = 0; for (let i = 0; i < count; i++) { if (at + 46 > cd.length || cd.readUInt32LE(at) !== CDH_SIG) { throw new ZipFormatError(`central directory entry ${i} is malformed`); } const flags = cd.readUInt16LE(at + 8); const method = cd.readUInt16LE(at + 10); const crc = cd.readUInt32LE(at + 16); let compressedSize = cd.readUInt32LE(at + 20); let size = cd.readUInt32LE(at + 24); const nameLen = cd.readUInt16LE(at + 28); const extraLen = cd.readUInt16LE(at + 30); const commentLen = cd.readUInt16LE(at + 32); let localHeaderOffset = cd.readUInt32LE(at + 42); const rawName = cd.subarray(at + 46, at + 46 + nameLen); const extras = parseExtras(cd.subarray(at + 46 + nameLen, at + 46 + nameLen + extraLen)); let name = decodeName(rawName, (flags & 0x800) !== 0); for (const ex of extras) { if (ex.id === 0x0001) { // ZIP64: only the fields whose 32-bit slot is saturated, in order. let p = 0; if (size === U32_MAX) { size = u64(ex.data, p); p += 8; } if (compressedSize === U32_MAX) { compressedSize = u64(ex.data, p); p += 8; } if (localHeaderOffset === U32_MAX) { localHeaderOffset = u64(ex.data, p); p += 8; } } else if (ex.id === 0x7075 && ex.data.length > 5 && ex.data[0] === 1) { // Info-ZIP Unicode Path: version 1, CRC32 of the raw name, UTF-8 name. if (ex.data.readUInt32LE(1) === crc32(rawName)) { name = ex.data.subarray(5).toString("utf8"); } } } name = name.replace(/\\/g, "/"); entries.push({ name, isDirectory: name.endsWith("/"), method, compressedSize, size, crc32: crc, localHeaderOffset, encrypted: (flags & 0x1) !== 0, }); at += 46 + nameLen + extraLen + commentLen; } return entries; } finally { await fh.close(); } } // Where an entry's data begins: past its local header, whose name and extra // lengths may differ from the central directory's. export async function zipEntryDataOffset(file: string, entry: ZipEntry): Promise { const fh = await open(file, "r"); try { const lfh = await readAt(fh, entry.localHeaderOffset, 30); if (lfh.length < 30 || lfh.readUInt32LE(0) !== LFH_SIG) { throw new ZipFormatError(`no local header for ${entry.name}`); } return entry.localHeaderOffset + 30 + lfh.readUInt16LE(26) + lfh.readUInt16LE(28); } finally { await fh.close(); } } // Why an entry cannot be copied out, or null when it can. export function zipEntryProblem(entry: ZipEntry): string | null { if (entry.isDirectory) return "a directory"; if (entry.encrypted) return "encrypted"; if (entry.method !== 0 && entry.method !== 8) { return `compression method ${entry.method} (only stored and deflated are read)`; } return null; } // Counts bytes, CRC32 and sha256 of what passes through it. class Digest extends Transform { bytes = 0; crc = 0; readonly hash = createHash("sha256"); _transform(chunk: Buffer, _enc: BufferEncoding, cb: TransformCallback): void { this.bytes += chunk.length; this.crc = crc32(chunk, this.crc); this.hash.update(chunk); cb(null, chunk); } } export type CopiedEntry = { bytes: number; sha256: string }; // Copy one entry's (uncompressed) bytes to `dest`, verifying its size and // CRC32 against the central directory. A stored entry is a byte range of the // archive; a deflated one is inflated. `dest` is written and left in place on // success; on any failure it is removed by the caller (it is a staging path). export async function copyZipEntry( file: string, entry: ZipEntry, dest: string, opts: { signal?: AbortSignal } = {}, ): Promise { const problem = zipEntryProblem(entry); if (problem) throw new ZipFormatError(`${entry.name}: ${problem}`); const start = await zipEntryDataOffset(file, entry); const digest = new Digest(); const out = createWriteStream(dest, { flags: "wx" }); if (entry.compressedSize === 0) { await pipeline(Readable.from([]), digest, out, { signal: opts.signal }); } else { const src = createReadStream(file, { start, end: start + entry.compressedSize - 1, highWaterMark: 1 << 20, }); if (entry.method === 0) { await pipeline(src, digest, out, { signal: opts.signal }); } else { await pipeline(src, createInflateRaw(), digest, out, { signal: opts.signal }); } } if (digest.bytes !== entry.size) { throw new ZipFormatError( `${entry.name}: ${digest.bytes} bytes copied, the archive says ${entry.size}`, ); } if (digest.crc >>> 0 !== entry.crc32 >>> 0) { throw new ZipFormatError(`${entry.name}: CRC32 mismatch — the archive is damaged here`); } return { bytes: digest.bytes, sha256: digest.hash.digest("hex") }; } // A small entry's text (an .info.json, a description.txt). Refuses anything // over `maxBytes` so a mislabelled video is never read into memory. export async function readZipEntryText( file: string, entry: ZipEntry, maxBytes = 4 * 1024 * 1024, ): Promise { const problem = zipEntryProblem(entry); if (problem) throw new ZipFormatError(`${entry.name}: ${problem}`); if (entry.size > maxBytes) { throw new ZipFormatError(`${entry.name}: ${entry.size} bytes is too large to read as text`); } const start = await zipEntryDataOffset(file, entry); const fh = await open(file, "r"); let raw: Buffer; try { raw = await readAt(fh, start, entry.compressedSize); } finally { await fh.close(); } if (entry.method === 8) { raw = inflateRawSync(raw); } return raw.toString("utf8"); } // Hash a plain file the same way (a directory source's entry). export async function copyFileHashed( src: string, dest: string, opts: { signal?: AbortSignal } = {}, ): Promise { const digest = new Digest(); await pipeline( createReadStream(src, { highWaterMark: 1 << 20 }), digest, createWriteStream(dest, { flags: "wx" }), { signal: opts.signal }, ); return { bytes: digest.bytes, sha256: digest.hash.digest("hex") }; }