/** * Streaming ZIP writer — store only, no compression. * * Written for downloading a whole directory out of a group. The archive can be * tens of gigabytes, so nothing is buffered: bytes go to the sink as they * arrive, and the only thing kept in memory is one small record per file for * the central directory at the end. * * Store-only is deliberate. What people put in a group is video, images and * archives — already compressed — so deflate would cost CPU on every byte to * save nothing, and it would have to run in the same thread that is decrypting * chunks. An uncompressed zip is also the one a stalled download leaves in a * recoverable state. * * Sizes are written after the data, in a data descriptor (general purpose bit * 3), because a stream cannot seek back to patch the header — the CRC is not * known until the last byte has gone past. Zip64 is used per entry when a file * is 4 GiB or larger, or when it starts past the 4 GiB mark, and for the * archive itself when it ends past that mark or holds more than 65535 files. * * No browser globals: this module is exercised under Node by * packages/meshbay-hub/tests/test_zipstream.py, which reads what it produces * with Python's zipfile and compares it byte for byte. */ const LOCAL_SIG = 0x04034b50; const DESC_SIG = 0x08074b50; const CENTRAL_SIG = 0x02014b50; const EOCD64_SIG = 0x06064b50; const LOC64_SIG = 0x07064b50; const EOCD_SIG = 0x06054b50; const U32_MAX = 0xffffffff; const ZIP64_THRESHOLD = 0xffffffff; // Bit 3: sizes and CRC follow the data. Bit 11: the name is UTF-8. const FLAG_DATA_DESCRIPTOR = 0x0008; const FLAG_UTF8 = 0x0800; const CRC_TABLE = (() => { const table = new Int32Array(256); for (let i = 0; i < 256; i++) { let c = i; for (let k = 0; k < 8; k++) c = c & 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1; table[i] = c; } return table; })(); export function crc32(bytes, seed = 0) { let c = ~seed; for (let i = 0; i < bytes.length; i++) { c = CRC_TABLE[(c ^ bytes[i]) & 0xff] ^ (c >>> 8); } return (~c) >>> 0; } /** MS-DOS time and date, which is what a zip entry carries. */ function dosDateTime(date) { const d = date instanceof Date ? date : new Date(date); const year = Math.max(1980, d.getFullYear()); return { time: (d.getHours() << 11) | (d.getMinutes() << 5) | (d.getSeconds() >> 1), date: ((year - 1980) << 9) | ((d.getMonth() + 1) << 5) | d.getDate(), }; } class Writer { constructor(size) { this.buf = new Uint8Array(size); this.view = new DataView(this.buf.buffer); this.off = 0; } u16(v) { this.view.setUint16(this.off, v, true); this.off += 2; return this; } u32(v) { this.view.setUint32(this.off, v >>> 0, true); this.off += 4; return this; } u64(v) { this.view.setBigUint64(this.off, BigInt(v), true); this.off += 8; return this; } bytes(b) { this.buf.set(b, this.off); this.off += b.length; return this; } } export class ZipStream { /** * @param {(bytes: Uint8Array) => Promise|void} sink where the archive goes */ constructor(sink) { this._sink = sink; this._offset = 0; // bytes written so far — every entry's offset this._entries = []; this._current = null; } async _write(bytes) { await this._sink(bytes); this._offset += bytes.length; } /** * Start a file. `size` is what the index says it will be; it decides whether * this entry needs zip64, and nothing is trusted about it afterwards — the * length actually written is what the archive records. */ async begin(name, size = 0, mtime = new Date(), { forceZip64 = false } = {}) { if (this._current) throw new Error('A file is already open in this archive'); const nameBytes = new TextEncoder().encode(name.replace(/\\/g, '/')); const zip64 = forceZip64 || size >= ZIP64_THRESHOLD || this._offset >= ZIP64_THRESHOLD; const { time, date } = dosDateTime(mtime); this._current = { name: nameBytes, zip64, time, date, offset: this._offset, crc: 0, size: 0, }; const extraLen = zip64 ? 20 : 0; const w = new Writer(30 + nameBytes.length + extraLen); w.u32(LOCAL_SIG) .u16(zip64 ? 45 : 20) .u16(FLAG_DATA_DESCRIPTOR | FLAG_UTF8) .u16(0) // stored .u16(time).u16(date) .u32(0) // crc — in the descriptor .u32(zip64 ? U32_MAX : 0) // compressed size .u32(zip64 ? U32_MAX : 0) // uncompressed size .u16(nameBytes.length) .u16(extraLen) .bytes(nameBytes); if (zip64) { // Placeholders: the real values go in the descriptor. The field has to be // here all the same, or a reader has no way to know the descriptor's // sizes are 8 bytes wide. w.u16(0x0001).u16(16).u64(0).u64(0); } await this._write(w.buf); } /** Feed the open file. Call as often as you like; nothing accumulates. */ async write(bytes) { if (!this._current) throw new Error('No file is open in this archive'); if (!bytes.length) return; this._current.crc = crc32(bytes, this._current.crc); this._current.size += bytes.length; await this._write(bytes); } /** Close the open file, writing what is now known about it. */ async end() { if (!this._current) throw new Error('No file is open in this archive'); const e = this._current; // A file that grew past 4 GiB after being announced smaller still has to be // described correctly, and its local header said otherwise. Recording it as // zip64 in the central directory is what readers go by. if (e.size >= ZIP64_THRESHOLD) e.zip64 = true; const w = new Writer(e.zip64 ? 24 : 16); w.u32(DESC_SIG).u32(e.crc); if (e.zip64) w.u64(e.size).u64(e.size); else w.u32(e.size).u32(e.size); await this._write(w.buf); this._entries.push(e); this._current = null; } /** Write the central directory and the end records. The archive is complete. */ async finish() { if (this._current) throw new Error('A file is still open in this archive'); const centralStart = this._offset; for (const e of this._entries) { const needsZip64 = e.zip64 || e.offset >= ZIP64_THRESHOLD; const extraLen = needsZip64 ? 32 : 0; const w = new Writer(46 + e.name.length + extraLen); w.u32(CENTRAL_SIG) .u16(0x031e) // made by: UNIX, spec 3.0 .u16(needsZip64 ? 45 : 20) .u16(FLAG_DATA_DESCRIPTOR | FLAG_UTF8) .u16(0) .u16(e.time).u16(e.date) .u32(e.crc) .u32(needsZip64 ? U32_MAX : e.size) .u32(needsZip64 ? U32_MAX : e.size) .u16(e.name.length) .u16(extraLen) .u16(0) // comment .u16(0) // disk .u16(0) // internal attrs .u32(0o644 << 16) // external attrs: rw-r--r-- .u32(needsZip64 ? U32_MAX : e.offset) .bytes(e.name); if (needsZip64) w.u16(0x0001).u16(28).u64(e.size).u64(e.size).u64(e.offset).u32(0); await this._write(w.buf); } const centralSize = this._offset - centralStart; const archiveZip64 = centralStart >= ZIP64_THRESHOLD || this._offset >= ZIP64_THRESHOLD || this._entries.length > 0xffff; if (archiveZip64) { const z = new Writer(56 + 20); z.u32(EOCD64_SIG).u64(44) // size of this record, less 12 .u16(0x031e).u16(45) .u32(0).u32(0) .u64(this._entries.length).u64(this._entries.length) .u64(centralSize).u64(centralStart) .u32(LOC64_SIG).u32(0).u64(centralStart + centralSize).u32(1); await this._write(z.buf); } const count = Math.min(this._entries.length, 0xffff); const w = new Writer(22); w.u32(EOCD_SIG).u16(0).u16(0).u16(count).u16(count) .u32(archiveZip64 ? U32_MAX : centralSize) .u32(archiveZip64 ? U32_MAX : centralStart) .u16(0); await this._write(w.buf); return this._offset; } } /** * Everything at or under `dir`, with the paths the archive should carry. * * `dir` is stripped from the front so an archive of "Holidays/2026" opens as * "2026/…" rather than as a chain of empty parents. */ export function entriesUnder(entries, dir) { const prefix = dir ? dir + '/' : ''; return entries .filter(e => (e.path || '') === dir || (e.path || '').startsWith(prefix)) .map(e => { const rest = (e.path || '').slice(dir.length).replace(/^\//, ''); const base = dir.split('/').pop() || 'files'; return { entry: e, name: [base, rest, e.name].filter(Boolean).join('/') }; }); }