crossbind
GitHub

zlib for WebAssembly

v1.3.2WebAssembly

zlib 1.3.2 for browsers, Node.js and edge runtimes, precompiled for wasm32, single-threaded and multi-threaded as @crossbind/port-zlib-wasm.

npm install @crossbind/port-zlib-wasm@beta

Install

shell
npm install @crossbind/port-zlib-wasm@beta
crossbind.config.js
import zlibWasm from '@crossbind/port-zlib-wasm/crossbind.config.js';
 
export default {
dependencies: [zlibWasm],
paths: { config: import.meta.url },
};

crossbind itself arrives with your bundler plugin, or with a new project from npm create crossbind@beta; Bundlers has Vite, Webpack, Rspack and Rollup.

Usage

Each example runs here, in this tab, and prints what the site build checked.

Each example also has a JavaScript only tab: the same task with no C++ file, calling zlib's own headers from @crossbind/port-zlib directly. 4 of 5 work that way; the other says what stops it.

Compress and decompress a buffer

The two calls most zlib code makes: compress2 at a chosen level, and uncompress with the original size the caller kept.

src/native/zlib_codec.h
#pragma once
 
#include <zlib.h>
 
#include <stdexcept>
#include <string>
 
// One-shot zlib: compress2 and uncompress, in the zlib format (RFC 1950). Bytes cross the binding as
// a byte string: one UTF-16 code unit (0-255) per byte.
class Zlib {
public:
static std::string version() { return zlibVersion(); }
 
static std::u16string compress(const std::string& text, int level) {
uLongf size = compressBound(static_cast<uLong>(text.size()));
std::string out(size, '\0');
const int status = compress2(reinterpret_cast<Bytef*>(&out[0]), &size, reinterpret_cast<const Bytef*>(text.data()), static_cast<uLong>(text.size()), level);
if (status != Z_OK) throw std::runtime_error(status == Z_STREAM_ERROR ? "level must be between -1 and 9" : zError(status));
std::u16string bytes(size, u'\0');
for (uLongf i = 0; i < size; ++i) bytes[i] = static_cast<unsigned char>(out[i]);
return bytes;
}
 
// The zlib format does not record the original size, so the caller keeps it next to the data.
static std::string decompress(const std::u16string& bytes, int originalSize) {
if (originalSize < 0 || originalSize > (256 << 20)) throw std::invalid_argument("originalSize must be between 0 and 256 MiB");
std::string in(bytes.size(), '\0');
for (size_t i = 0; i < bytes.size(); ++i) {
if (bytes[i] > 0xFF) throw std::invalid_argument("not a byte string");
in[i] = static_cast<char>(bytes[i]);
}
std::string out(static_cast<size_t>(originalSize), '\0');
uLongf size = static_cast<uLongf>(out.size());
const int status = uncompress(reinterpret_cast<Bytef*>(&out[0]), &size, reinterpret_cast<const Bytef*>(in.data()), static_cast<uLong>(in.size()));
if (status == Z_BUF_ERROR) throw std::runtime_error("the data is larger than originalSize");
if (status != Z_OK) throw std::runtime_error(status == Z_DATA_ERROR ? "corrupt or incomplete zlib data" : zError(status));
out.resize(size);
return out;
}
};
main.js
import { initNative, Zlib } from './native/zlib_codec.h';
 
await initNative();
const text = 'crossbind '.repeat(100);
const packed = await Zlib.compress(text, 9);
const bytes = Uint8Array.from(packed, (c) => c.charCodeAt(0));
console.log(await Zlib.version(), bytes.length, [...bytes.slice(0, 2)].map((b) => b.toString(16)).join(' '));
console.log((await Zlib.decompress(packed, text.length)) === text);
PRINTSfirst run downloads 0.5 MB
1.3.2 27 78 da
true

Write a .gz that remembers its file name

deflateInit2 with windowBits 15 + 16 writes gzip instead of zlib, and deflateSetHeader fills the header fields that gunzip -N restores. inflateGetHeader reads them back.

src/native/gzip_header.h
#pragma once
 
#include <zlib.h>
 
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// gzip (RFC 1952) with the header fields gunzip -N restores: the original file name and the
// modification time. Bytes cross the binding as a byte string: one UTF-16 code unit (0-255) per byte.
class Gzip {
public:
static std::u16string compress(const std::string& data, const std::string& name, double mtime, int level) {
z_stream stream{};
// windowBits 15 + 16: a 32 KiB window, and the gzip wrapper instead of the zlib one.
if (deflateInit2(&stream, level, Z_DEFLATED, 15 + 16, 8, Z_DEFAULT_STRATEGY) != Z_OK) throw std::invalid_argument("level must be between -1 and 9");
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, deflateEnd);
gz_header header{};
header.name = reinterpret_cast<Bytef*>(const_cast<char*>(name.c_str()));
header.time = static_cast<uLong>(mtime);
header.os = 3; // Unix
deflateSetHeader(&stream, &header);
std::string out(deflateBound(&stream, static_cast<uLong>(data.size())), '\0');
stream.next_in = reinterpret_cast<Bytef*>(const_cast<char*>(data.data()));
stream.avail_in = static_cast<uInt>(data.size());
stream.next_out = reinterpret_cast<Bytef*>(&out[0]);
stream.avail_out = static_cast<uInt>(out.size());
if (deflate(&stream, Z_FINISH) != Z_STREAM_END) throw std::runtime_error("deflate did not finish");
std::u16string bytes(stream.total_out, u'\0');
for (size_t i = 0; i < bytes.size(); ++i) bytes[i] = static_cast<unsigned char>(out[i]);
return bytes;
}
 
static std::string fileName(const std::u16string& gz) {
std::string name;
uLong mtime = 0;
readHeader(gz, name, mtime);
return name;
}
 
// Seconds since 1970-01-01T00:00:00Z.
static double modified(const std::u16string& gz) {
std::string name;
uLong mtime = 0;
readHeader(gz, name, mtime);
return static_cast<double>(mtime);
}
 
static std::string decompress(const std::u16string& gz) {
std::string in = fromUnits(gz);
z_stream stream{};
if (inflateInit2(&stream, 15 + 32) != Z_OK) throw std::runtime_error("inflateInit2 failed"); // + 32: gzip or zlib, from the header
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, inflateEnd);
stream.next_in = reinterpret_cast<Bytef*>(&in[0]);
stream.avail_in = static_cast<uInt>(in.size());
std::string out;
std::vector<char> chunk(64 << 10);
int status = Z_OK;
while (status != Z_STREAM_END) {
stream.next_out = reinterpret_cast<Bytef*>(chunk.data());
stream.avail_out = static_cast<uInt>(chunk.size());
status = inflate(&stream, Z_NO_FLUSH);
if (status != Z_OK && status != Z_STREAM_END) throw std::runtime_error(stream.msg ? stream.msg : "truncated or corrupt input");
out.append(chunk.data(), chunk.size() - stream.avail_out);
if (out.size() > (256u << 20)) throw std::runtime_error("refusing to decompress more than 256 MiB");
}
return out;
}
 
private:
static std::string fromUnits(const std::u16string& units) {
std::string bytes(units.size(), '\0');
for (size_t i = 0; i < units.size(); ++i) {
if (units[i] > 0xFF) throw std::invalid_argument("not a byte string");
bytes[i] = static_cast<char>(units[i]);
}
return bytes;
}
 
// inflateGetHeader fills the fields while inflate reads the header; Z_BLOCK stops right after it.
static void readHeader(const std::u16string& gz, std::string& name, uLong& mtime) {
std::string in = fromUnits(gz);
z_stream stream{};
if (inflateInit2(&stream, 15 + 16) != Z_OK) throw std::runtime_error("inflateInit2 failed"); // + 16: gzip only
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, inflateEnd);
std::vector<Bytef> nameBuffer(1024, 0);
gz_header header{};
header.name = nameBuffer.data();
header.name_max = static_cast<uInt>(nameBuffer.size() - 1);
inflateGetHeader(&stream, &header);
Bytef output = 0;
stream.next_in = reinterpret_cast<Bytef*>(&in[0]);
stream.avail_in = static_cast<uInt>(in.size());
stream.next_out = &output;
stream.avail_out = 1;
const int status = inflate(&stream, Z_BLOCK);
if ((status != Z_OK && status != Z_STREAM_END) || header.done != 1) throw std::runtime_error(stream.msg ? stream.msg : "not a complete gzip header");
name = reinterpret_cast<const char*>(nameBuffer.data());
mtime = header.time;
}
};
main.js
import { initNative, Gzip } from './native/gzip_header.h';
 
await initNative();
const rows = Array.from({ length: 200 }, (_, i) => `${i + 1},sensor-${i % 4},${18 + ((i * 7) % 9)}`);
const csv = ['reading,sensor,celsius', ...rows].join('\n') + '\n';
const gz = await Gzip.compress(csv, 'readings.csv', Date.UTC(2026, 0, 1) / 1000, 9);
const bytes = Uint8Array.from(gz, (c) => c.charCodeAt(0));
console.log(`${csv.length} B -> ${bytes.length} B, starts ${bytes[0].toString(16)} ${bytes[1].toString(16)}`);
console.log(await Gzip.fileName(gz), new Date((await Gzip.modified(gz)) * 1000).toISOString());
console.log((await Gzip.decompress(gz)) === csv);
PRINTSfirst run downloads 0.5 MB
3115 B -> 747 B, starts 1f 8b
readings.csv 2026-01-01T00:00:00.000Z
true

Stream a file through gzip

For data you should not hold in one buffer: the deflate() and inflate() loops of zpipe.c, the example that ships with zlib, work file to file in 64 KB steps.

src/native/gzip_stream.h
#pragma once
 
#include <zlib.h>
 
#include <cstdio>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// Streaming gzip, file to file, in 64 KiB steps: the deflate() and inflate() loops of zlib's own
// examples/zpipe.c, with windowBits 15 + 16 for gzip. Memory stays at two buffers whatever the size.
class GzipStream {
public:
// Returns the compressed size.
static double compressFile(const std::string& input, const std::string& output, int level) {
File in = open(input, "rb");
File out = open(output, "wb");
z_stream stream{};
if (deflateInit2(&stream, level, Z_DEFLATED, 15 + 16, 8, Z_DEFAULT_STRATEGY) != Z_OK) throw std::invalid_argument("level must be between -1 and 9");
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, deflateEnd);
std::vector<unsigned char> source(CHUNK);
std::vector<unsigned char> target(CHUNK);
double written = 0;
int flush = Z_NO_FLUSH;
do {
stream.avail_in = static_cast<uInt>(std::fread(source.data(), 1, CHUNK, in.get()));
if (std::ferror(in.get())) throw std::runtime_error("cannot read " + input);
flush = std::feof(in.get()) ? Z_FINISH : Z_NO_FLUSH;
stream.next_in = source.data();
do {
stream.avail_out = CHUNK;
stream.next_out = target.data();
deflate(&stream, flush);
written += write(out.get(), target.data(), CHUNK - stream.avail_out);
} while (stream.avail_out == 0);
} while (flush != Z_FINISH);
return written;
}
 
// Returns the decompressed size. Reads every member of a concatenated .gz, as gunzip does.
static double decompressFile(const std::string& input, const std::string& output) {
File in = open(input, "rb");
File out = open(output, "wb");
z_stream stream{};
if (inflateInit2(&stream, 15 + 32) != Z_OK) throw std::runtime_error("inflateInit2 failed"); // + 32: gzip or zlib
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, inflateEnd);
std::vector<unsigned char> source(CHUNK);
std::vector<unsigned char> target(CHUNK);
double written = 0;
int status = Z_OK;
for (;;) {
stream.avail_in = static_cast<uInt>(std::fread(source.data(), 1, CHUNK, in.get()));
if (std::ferror(in.get())) throw std::runtime_error("cannot read " + input);
if (stream.avail_in == 0) break;
stream.next_in = source.data();
do {
if (status == Z_STREAM_END) {
if (stream.avail_in == 0) break;
inflateReset(&stream); // another member follows
}
stream.avail_out = CHUNK;
stream.next_out = target.data();
status = inflate(&stream, Z_NO_FLUSH);
if (status == Z_NEED_DICT || status == Z_DATA_ERROR || status == Z_MEM_ERROR) throw std::runtime_error(stream.msg ? stream.msg : "corrupt input");
written += write(out.get(), target.data(), CHUNK - stream.avail_out);
} while (stream.avail_out == 0 || (status == Z_STREAM_END && stream.avail_in > 0));
}
if (status != Z_STREAM_END) throw std::runtime_error("the input ends in the middle of a gzip member");
return written;
}
 
private:
static constexpr size_t CHUNK = 64 << 10;
using File = std::unique_ptr<FILE, int (*)(FILE*)>;
 
static File open(const std::string& path, const char* mode) {
File file(std::fopen(path.c_str(), mode), std::fclose);
if (!file) throw std::runtime_error("cannot open " + path);
return file;
}
 
static double write(FILE* file, const unsigned char* data, size_t size) {
if (size && std::fwrite(data, 1, size, file) != size) throw std::runtime_error("write failed");
return static_cast<double>(size);
}
};
main.js
import { initNative } from './native/gzip_stream.h';
 
const m = await initNative();
const { GzipStream } = m;
let seed = 42;
const random = (n) => (seed = (seed * 48271) % 2147483647) % n;
const lines = Array.from({ length: 50000 }, (_, i) => `2026-09-24T12:00:${String(i % 60).padStart(2, '0')}Z GET /api/items/${random(9000)} ${random(10) ? 200 : 404} ${random(900)}ms`);
// m.FS.writeFile adds to a file that already exists, so every run gets a fresh directory.
const dir = await m.getRandomPath('/memfs');
await m.FS.writeFile(`${dir}/access.log`, lines.join('\n'));
 
const packed = await GzipStream.compressFile(`${dir}/access.log`, `${dir}/access.log.gz`, 6);
const unpacked = await GzipStream.decompressFile(`${dir}/access.log.gz`, `${dir}/access.copy.log`);
const original = await m.getFileBytes(`${dir}/access.log`);
const copy = await m.getFileBytes(`${dir}/access.copy.log`);
console.log(`${original.length} B -> ${packed} B -> ${unpacked} B`);
console.log(copy.length === original.length && copy.every((byte, i) => byte === original[i]));
PRINTSfirst run downloads 0.5 MB
2537578 B -> 363636 B -> 2537578 B
true

Compress small messages with a preset dictionary

A short message has little to refer back to. Load earlier messages with deflateSetDictionary and inflateSetDictionary, and each new one compresses against them.

src/native/zlib_dictionary.h
#pragma once
 
#include <zlib.h>
 
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// Raw deflate (RFC 1951, no header or checksum) with a preset dictionary: both sides load the same
// bytes first, so a short message can point back at strings it never contained. An empty dictionary
// gives plain raw deflate.
class ZlibDictionary {
public:
ZlibDictionary(const std::string& dictionary, int level) : dictionary(dictionary), level(level) {}
 
std::u16string compress(const std::string& message) const {
z_stream stream{};
// windowBits -15: raw deflate with a 32 KiB window.
if (deflateInit2(&stream, level, Z_DEFLATED, -15, 8, Z_DEFAULT_STRATEGY) != Z_OK) throw std::invalid_argument("level must be between -1 and 9");
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, deflateEnd);
if (!dictionary.empty()) deflateSetDictionary(&stream, reinterpret_cast<const Bytef*>(dictionary.data()), static_cast<uInt>(dictionary.size()));
std::string out(deflateBound(&stream, static_cast<uLong>(message.size())), '\0');
stream.next_in = reinterpret_cast<Bytef*>(const_cast<char*>(message.data()));
stream.avail_in = static_cast<uInt>(message.size());
stream.next_out = reinterpret_cast<Bytef*>(&out[0]);
stream.avail_out = static_cast<uInt>(out.size());
if (deflate(&stream, Z_FINISH) != Z_STREAM_END) throw std::runtime_error("deflate did not finish");
std::u16string bytes(stream.total_out, u'\0');
for (size_t i = 0; i < bytes.size(); ++i) bytes[i] = static_cast<unsigned char>(out[i]);
return bytes;
}
 
std::string decompress(const std::u16string& bytes) const {
std::string in(bytes.size(), '\0');
for (size_t i = 0; i < bytes.size(); ++i) {
if (bytes[i] > 0xFF) throw std::invalid_argument("not a byte string");
in[i] = static_cast<char>(bytes[i]);
}
z_stream stream{};
if (inflateInit2(&stream, -15) != Z_OK) throw std::runtime_error("inflateInit2 failed");
std::unique_ptr<z_stream, int (*)(z_stream*)> end(&stream, inflateEnd);
// Raw deflate carries no dictionary ID, so the dictionary goes in before the first inflate().
if (!dictionary.empty()) inflateSetDictionary(&stream, reinterpret_cast<const Bytef*>(dictionary.data()), static_cast<uInt>(dictionary.size()));
stream.next_in = reinterpret_cast<Bytef*>(&in[0]);
stream.avail_in = static_cast<uInt>(in.size());
std::string out;
std::vector<char> chunk(16 << 10);
int status = Z_OK;
while (status != Z_STREAM_END) {
stream.next_out = reinterpret_cast<Bytef*>(chunk.data());
stream.avail_out = static_cast<uInt>(chunk.size());
status = inflate(&stream, Z_NO_FLUSH);
if (status != Z_OK && status != Z_STREAM_END) throw std::runtime_error(stream.msg ? stream.msg : "truncated or corrupt input");
out.append(chunk.data(), chunk.size() - stream.avail_out);
if (out.size() > (1u << 20)) throw std::runtime_error("a message above 1 MiB is not a small message");
}
return out;
}
 
private:
std::string dictionary;
int level;
};
main.js
import { initNative, ZlibDictionary } from './native/zlib_dictionary.h';
 
await initNative();
const event = (i) => `{"event":"click","user":${1000 + ((i * 37) % 900)},"page":"/products/${i % 12}","ms":${(i * 7919) % 400}}`;
const dictionary = Array.from({ length: 20 }, (_, i) => event(i)).join('\n');
const plain = await new ZlibDictionary('', 9);
const primed = await new ZlibDictionary(dictionary, 9);
 
const message = event(4321);
const alone = await plain.compress(message);
const packed = await primed.compress(message);
console.log(`dictionary: ${dictionary.length} B`);
console.log(`${message.length} B message: ${alone.length} B alone, ${packed.length} B with the dictionary`);
console.log((await primed.decompress(packed)) === message);
PRINTSfirst run downloads 0.5 MB
dictionary: 1195 B
59 B message: 59 B alone, 11 B with the dictionary
true

Checksum data in pieces

crc32 and adler32 continue a running value across pieces, and crc32_combine and adler32_combine join the checksums of pieces computed separately.

src/native/zlib_checksum.h
#pragma once
 
#include <zlib.h>
 
#include <string>
 
// CRC-32 (what gzip, ZIP and PNG store) and Adler-32 (what the zlib format stores). A running value
// continues across pieces, and two values combine into the checksum of the joined data, so pieces can
// be checksummed separately, in any order.
class Checksum {
public:
static unsigned int crc32(const std::string& data) { return crc32Update(0, data); }
static unsigned int adler32(const std::string& data) { return adler32Update(1, data); }
 
static unsigned int crc32Update(unsigned int crc, const std::string& data) {
return static_cast<unsigned int>(::crc32(crc, reinterpret_cast<const Bytef*>(data.data()), static_cast<uInt>(data.size())));
}
 
static unsigned int adler32Update(unsigned int adler, const std::string& data) {
return static_cast<unsigned int>(::adler32(adler, reinterpret_cast<const Bytef*>(data.data()), static_cast<uInt>(data.size())));
}
 
// The checksum of first + second, from the two checksums and the length of second in bytes.
static unsigned int crc32Combine(unsigned int first, unsigned int second, double secondLength) {
return static_cast<unsigned int>(::crc32_combine(first, second, static_cast<z_off_t>(secondLength)));
}
 
static unsigned int adler32Combine(unsigned int first, unsigned int second, double secondLength) {
return static_cast<unsigned int>(::adler32_combine(first, second, static_cast<z_off_t>(secondLength)));
}
};
main.js
import { initNative, Checksum } from './native/zlib_checksum.h';
 
await initNative();
const hex = (value) => value.toString(16).padStart(8, '0');
const text = 'The quick brown fox jumps over the lazy dog';
console.log(`crc32 ${hex(await Checksum.crc32(text))}, adler32 ${hex(await Checksum.adler32(text))}`);
 
const [head, tail] = ['The quick brown fox ', 'jumps over the lazy dog'];
const running = await Checksum.crc32Update(await Checksum.crc32(head), tail);
const combined = await Checksum.crc32Combine(await Checksum.crc32(head), await Checksum.crc32(tail), tail.length);
const adler = await Checksum.adler32Combine(await Checksum.adler32(head), await Checksum.adler32(tail), tail.length);
console.log(`running crc32 ${hex(running)}, combined crc32 ${hex(combined)}, combined adler32 ${hex(adler)}`);
PRINTSfirst run downloads 0.5 MB
crc32 414fa339, adler32 5bdc0fda
running crc32 414fa339, combined crc32 414fa339, combined adler32 5bdc0fda

What is different on WebAssembly

  • In a browser the module runs in a Worker by default (useWorker), so every call returns a promise: await calls and constructors alike.
  • The module has its own filesystem: m.FS writes files, m.getFileBytes reads them back and m.autoMountFiles mounts File objects from an <input type=file>. /memfs lives in memory; /opfs persists across reloads and needs the Worker. See Filesystem.
  • In Node.js, m.FS is the real disk, so use real paths there.
  • Multi-threaded builds (runtime: 'mt') need COOP and COEP headers in production. See Threading.

Other platforms

Facts on this page come from the port manifests in the repository and from what npm served on beta when the site was built. See the Libraries guide for the full consumer flow.

MORE LIBRARIES
cURLExpatGDALGEOSGeoTIFFiconvLERClibjpeg-turbolibTIFFOpenSSLPROJSpatiaLiteSQLiteWebPZstandard
Type to search every guide page and section.
↑↓ navigate↵ openesc close