summaryrefslogtreecommitdiffstats
path: root/packages/meshbay-hub/src/meshbay_hub/static/zipstream.js
diff options
context:
space:
mode:
Diffstat (limited to 'packages/meshbay-hub/src/meshbay_hub/static/zipstream.js')
-rw-r--r--packages/meshbay-hub/src/meshbay_hub/static/zipstream.js241
1 files changed, 241 insertions, 0 deletions
diff --git a/packages/meshbay-hub/src/meshbay_hub/static/zipstream.js b/packages/meshbay-hub/src/meshbay_hub/static/zipstream.js
new file mode 100644
index 0000000..bf2230a
--- /dev/null
+++ b/packages/meshbay-hub/src/meshbay_hub/static/zipstream.js
@@ -0,0 +1,241 @@
+/**
+ * Streaming ZIP writer — store only, no compression.
+ *
+ * Written for downloading a whole directory out of a group. The archive can be
+ * tens of gigabytes, so nothing is buffered: bytes go to the sink as they
+ * arrive, and the only thing kept in memory is one small record per file for
+ * the central directory at the end.
+ *
+ * Store-only is deliberate. What people put in a group is video, images and
+ * archives — already compressed — so deflate would cost CPU on every byte to
+ * save nothing, and it would have to run in the same thread that is decrypting
+ * chunks. An uncompressed zip is also the one a stalled download leaves in a
+ * recoverable state.
+ *
+ * Sizes are written after the data, in a data descriptor (general purpose bit
+ * 3), because a stream cannot seek back to patch the header — the CRC is not
+ * known until the last byte has gone past. Zip64 is used per entry when a file
+ * is 4 GiB or larger, or when it starts past the 4 GiB mark, and for the
+ * archive itself when it ends past that mark or holds more than 65535 files.
+ *
+ * No browser globals: this module is exercised under Node by
+ * packages/meshbay-hub/tests/test_zipstream.py, which reads what it produces
+ * with Python's zipfile and compares it byte for byte.
+ */
+
+const LOCAL_SIG = 0x04034b50;
+const DESC_SIG = 0x08074b50;
+const CENTRAL_SIG = 0x02014b50;
+const EOCD64_SIG = 0x06064b50;
+const LOC64_SIG = 0x07064b50;
+const EOCD_SIG = 0x06054b50;
+
+const U32_MAX = 0xffffffff;
+const ZIP64_THRESHOLD = 0xffffffff;
+
+// Bit 3: sizes and CRC follow the data. Bit 11: the name is UTF-8.
+const FLAG_DATA_DESCRIPTOR = 0x0008;
+const FLAG_UTF8 = 0x0800;
+
+const CRC_TABLE = (() => {
+ const table = new Int32Array(256);
+ for (let i = 0; i < 256; i++) {
+ let c = i;
+ for (let k = 0; k < 8; k++) c = c & 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1;
+ table[i] = c;
+ }
+ return table;
+})();
+
+export function crc32(bytes, seed = 0) {
+ let c = ~seed;
+ for (let i = 0; i < bytes.length; i++) {
+ c = CRC_TABLE[(c ^ bytes[i]) & 0xff] ^ (c >>> 8);
+ }
+ return (~c) >>> 0;
+}
+
+/** MS-DOS time and date, which is what a zip entry carries. */
+function dosDateTime(date) {
+ const d = date instanceof Date ? date : new Date(date);
+ const year = Math.max(1980, d.getFullYear());
+ return {
+ time: (d.getHours() << 11) | (d.getMinutes() << 5) | (d.getSeconds() >> 1),
+ date: ((year - 1980) << 9) | ((d.getMonth() + 1) << 5) | d.getDate(),
+ };
+}
+
+class Writer {
+ constructor(size) {
+ this.buf = new Uint8Array(size);
+ this.view = new DataView(this.buf.buffer);
+ this.off = 0;
+ }
+ u16(v) { this.view.setUint16(this.off, v, true); this.off += 2; return this; }
+ u32(v) { this.view.setUint32(this.off, v >>> 0, true); this.off += 4; return this; }
+ u64(v) {
+ this.view.setBigUint64(this.off, BigInt(v), true);
+ this.off += 8;
+ return this;
+ }
+ bytes(b) { this.buf.set(b, this.off); this.off += b.length; return this; }
+}
+
+export class ZipStream {
+ /**
+ * @param {(bytes: Uint8Array) => Promise<void>|void} sink where the archive goes
+ */
+ constructor(sink) {
+ this._sink = sink;
+ this._offset = 0; // bytes written so far — every entry's offset
+ this._entries = [];
+ this._current = null;
+ }
+
+ async _write(bytes) {
+ await this._sink(bytes);
+ this._offset += bytes.length;
+ }
+
+ /**
+ * Start a file. `size` is what the index says it will be; it decides whether
+ * this entry needs zip64, and nothing is trusted about it afterwards — the
+ * length actually written is what the archive records.
+ */
+ async begin(name, size = 0, mtime = new Date(), { forceZip64 = false } = {}) {
+ if (this._current) throw new Error('A file is already open in this archive');
+
+ const nameBytes = new TextEncoder().encode(name.replace(/\\/g, '/'));
+ const zip64 = forceZip64
+ || size >= ZIP64_THRESHOLD
+ || this._offset >= ZIP64_THRESHOLD;
+ const { time, date } = dosDateTime(mtime);
+
+ this._current = {
+ name: nameBytes, zip64, time, date,
+ offset: this._offset, crc: 0, size: 0,
+ };
+
+ const extraLen = zip64 ? 20 : 0;
+ const w = new Writer(30 + nameBytes.length + extraLen);
+ w.u32(LOCAL_SIG)
+ .u16(zip64 ? 45 : 20)
+ .u16(FLAG_DATA_DESCRIPTOR | FLAG_UTF8)
+ .u16(0) // stored
+ .u16(time).u16(date)
+ .u32(0) // crc — in the descriptor
+ .u32(zip64 ? U32_MAX : 0) // compressed size
+ .u32(zip64 ? U32_MAX : 0) // uncompressed size
+ .u16(nameBytes.length)
+ .u16(extraLen)
+ .bytes(nameBytes);
+ if (zip64) {
+ // Placeholders: the real values go in the descriptor. The field has to be
+ // here all the same, or a reader has no way to know the descriptor's
+ // sizes are 8 bytes wide.
+ w.u16(0x0001).u16(16).u64(0).u64(0);
+ }
+ await this._write(w.buf);
+ }
+
+ /** Feed the open file. Call as often as you like; nothing accumulates. */
+ async write(bytes) {
+ if (!this._current) throw new Error('No file is open in this archive');
+ if (!bytes.length) return;
+ this._current.crc = crc32(bytes, this._current.crc);
+ this._current.size += bytes.length;
+ await this._write(bytes);
+ }
+
+ /** Close the open file, writing what is now known about it. */
+ async end() {
+ if (!this._current) throw new Error('No file is open in this archive');
+ const e = this._current;
+ // A file that grew past 4 GiB after being announced smaller still has to be
+ // described correctly, and its local header said otherwise. Recording it as
+ // zip64 in the central directory is what readers go by.
+ if (e.size >= ZIP64_THRESHOLD) e.zip64 = true;
+
+ const w = new Writer(e.zip64 ? 24 : 16);
+ w.u32(DESC_SIG).u32(e.crc);
+ if (e.zip64) w.u64(e.size).u64(e.size);
+ else w.u32(e.size).u32(e.size);
+ await this._write(w.buf);
+
+ this._entries.push(e);
+ this._current = null;
+ }
+
+ /** Write the central directory and the end records. The archive is complete. */
+ async finish() {
+ if (this._current) throw new Error('A file is still open in this archive');
+
+ const centralStart = this._offset;
+ for (const e of this._entries) {
+ const needsZip64 = e.zip64 || e.offset >= ZIP64_THRESHOLD;
+ const extraLen = needsZip64 ? 32 : 0;
+ const w = new Writer(46 + e.name.length + extraLen);
+ w.u32(CENTRAL_SIG)
+ .u16(0x031e) // made by: UNIX, spec 3.0
+ .u16(needsZip64 ? 45 : 20)
+ .u16(FLAG_DATA_DESCRIPTOR | FLAG_UTF8)
+ .u16(0)
+ .u16(e.time).u16(e.date)
+ .u32(e.crc)
+ .u32(needsZip64 ? U32_MAX : e.size)
+ .u32(needsZip64 ? U32_MAX : e.size)
+ .u16(e.name.length)
+ .u16(extraLen)
+ .u16(0) // comment
+ .u16(0) // disk
+ .u16(0) // internal attrs
+ .u32(0o644 << 16) // external attrs: rw-r--r--
+ .u32(needsZip64 ? U32_MAX : e.offset)
+ .bytes(e.name);
+ if (needsZip64) w.u16(0x0001).u16(28).u64(e.size).u64(e.size).u64(e.offset).u32(0);
+ await this._write(w.buf);
+ }
+ const centralSize = this._offset - centralStart;
+
+ const archiveZip64 = centralStart >= ZIP64_THRESHOLD
+ || this._offset >= ZIP64_THRESHOLD
+ || this._entries.length > 0xffff;
+
+ if (archiveZip64) {
+ const z = new Writer(56 + 20);
+ z.u32(EOCD64_SIG).u64(44) // size of this record, less 12
+ .u16(0x031e).u16(45)
+ .u32(0).u32(0)
+ .u64(this._entries.length).u64(this._entries.length)
+ .u64(centralSize).u64(centralStart)
+ .u32(LOC64_SIG).u32(0).u64(centralStart + centralSize).u32(1);
+ await this._write(z.buf);
+ }
+
+ const count = Math.min(this._entries.length, 0xffff);
+ const w = new Writer(22);
+ w.u32(EOCD_SIG).u16(0).u16(0).u16(count).u16(count)
+ .u32(archiveZip64 ? U32_MAX : centralSize)
+ .u32(archiveZip64 ? U32_MAX : centralStart)
+ .u16(0);
+ await this._write(w.buf);
+ return this._offset;
+ }
+}
+
+/**
+ * Everything at or under `dir`, with the paths the archive should carry.
+ *
+ * `dir` is stripped from the front so an archive of "Holidays/2026" opens as
+ * "2026/…" rather than as a chain of empty parents.
+ */
+export function entriesUnder(entries, dir) {
+ const prefix = dir ? dir + '/' : '';
+ return entries
+ .filter(e => (e.path || '') === dir || (e.path || '').startsWith(prefix))
+ .map(e => {
+ const rest = (e.path || '').slice(dir.length).replace(/^\//, '');
+ const base = dir.split('/').pop() || 'files';
+ return { entry: e, name: [base, rest, e.name].filter(Boolean).join('/') };
+ });
+}