crossbind
GitHub

Expat for WebAssembly

v2.8.5WebAssembly

Expat 2.8.5 for browsers, Node.js and edge runtimes, precompiled for wasm32, single-threaded and multi-threaded as @crossbind/port-expat-wasm.

npm install @crossbind/port-expat-wasm@beta

Install

shell
npm install @crossbind/port-expat-wasm@beta
crossbind.config.js
import expatWasm from '@crossbind/port-expat-wasm/crossbind.config.js';
 
export default {
dependencies: [expatWasm],
paths: { config: import.meta.url },
};

crossbind itself arrives with your bundler plugin, or with a new project from npm create crossbind@beta; Bundlers has Vite, Webpack, Rspack and Rollup.

Usage

Each example runs here, in this tab, and prints what the site build checked.

Each example also has a JavaScript only tab: the same task with no C++ file, calling Expat's own headers from @crossbind/port-expat directly. All 4 work that way.

Turn XML into JavaScript objects

The calls most Expat code makes: XML_ParserCreate, an element handler and a character data handler, then XML_Parse. A malformed document throws with the line and column where Expat stopped.

src/native/xml_tree.h
#pragma once
 
#include <expat.h>
 
#include <stdexcept>
#include <string>
#include <vector>
 
// XML in, JSON out. Each element becomes {"name","attributes","children","text"}: `children` holds
// its child elements and `text` its own character data, left out when it is only whitespace.
// Malformed XML throws Expat's message with the line and column where parsing stopped.
class XmlTree {
public:
static std::string version() {
const XML_Expat_Version v = XML_ExpatVersionInfo();
return std::to_string(v.major) + "." + std::to_string(v.minor) + "." + std::to_string(v.micro);
}
 
static std::string parse(const std::string& xml) {
Builder builder;
XML_Parser parser = XML_ParserCreate(nullptr);
if (!parser) throw std::runtime_error("out of memory");
XML_SetUserData(parser, &builder);
XML_SetElementHandler(parser, onStart, onEnd);
XML_SetCharacterDataHandler(parser, onText);
const bool ok = XML_Parse(parser, xml.data(), static_cast<int>(xml.size()), XML_TRUE) == XML_STATUS_OK;
const std::string error = ok ? "" : std::string(XML_ErrorString(XML_GetErrorCode(parser))) + " at line " +
std::to_string(XML_GetCurrentLineNumber(parser)) + ", column " +
std::to_string(XML_GetCurrentColumnNumber(parser));
XML_ParserFree(parser);
if (!ok) throw std::runtime_error(error);
return builder.json;
}
 
private:
struct Builder {
std::string json;
std::vector<std::string> text; // character data of each open element
std::vector<bool> hasChildren;
};
 
// Expat delivers UTF-8; JSON only needs quotes, backslashes and control characters escaped.
static std::string quote(const std::string& value) {
std::string out = "\"";
for (const char c : value) {
if (c == '"' || c == '\\') {
out += '\\';
out += c;
} else if (c == '\n') {
out += "\\n";
} else if (c == '\t') {
out += "\\t";
} else if (c == '\r') {
out += "\\r";
} else {
out += c;
}
}
return out + "\"";
}
 
static void XMLCALL onStart(void* data, const XML_Char* name, const XML_Char** attributes) {
Builder& builder = *static_cast<Builder*>(data);
if (!builder.hasChildren.empty()) {
if (builder.hasChildren.back()) builder.json += ",";
builder.hasChildren.back() = true;
}
builder.json += "{\"name\":" + quote(name) + ",\"attributes\":{";
for (int i = 0; attributes[i]; i += 2) builder.json += (i ? "," : "") + quote(attributes[i]) + ":" + quote(attributes[i + 1]);
builder.json += "},\"children\":[";
builder.text.emplace_back();
builder.hasChildren.push_back(false);
}
 
static void XMLCALL onText(void* data, const XML_Char* text, int length) {
static_cast<Builder*>(data)->text.back().append(text, static_cast<size_t>(length));
}
 
static void XMLCALL onEnd(void* data, const XML_Char*) {
Builder& builder = *static_cast<Builder*>(data);
const std::string& text = builder.text.back();
builder.json += "]";
if (text.find_first_not_of(" \t\r\n") != std::string::npos) builder.json += ",\"text\":" + quote(text);
builder.json += "}";
builder.text.pop_back();
builder.hasChildren.pop_back();
}
};
main.js
import { initNative, XmlTree } from './native/xml_tree.h';
 
await initNative();
console.log('Expat', await XmlTree.version());
const xml = `<?xml version="1.0" encoding="UTF-8"?>
<catalog>
<book id="1" lang="en"><title>Dune</title><price currency="EUR">9.99</price></book>
<book id="2" lang="fr"><title>L'Étranger</title><price currency="EUR">7.50</price></book>
<book id="3" lang="en"><title>Pride &amp; Prejudice</title><price currency="GBP">5.25</price></book>
</catalog>`;
const catalog = JSON.parse(await XmlTree.parse(xml));
for (const book of catalog.children) {
const [title, price] = book.children;
console.log(book.attributes.id, book.attributes.lang, title.text, price.text, price.attributes.currency);
}
 
try {
await XmlTree.parse('<catalog>\n <book id="4"><title>Emma</book>\n</catalog>');
} catch (error) {
console.log(error.cppMessage ?? error.message);
}
PRINTSfirst run downloads 0.6 MB
Expat 2.8.5
1 en Dune 9.99 EUR
2 fr L'Étranger 7.50 EUR
3 en Pride & Prejudice 5.25 GBP
mismatched tag at line 2, column 30

Stream a file through Expat

For documents you should not hold in one string: XML_GetBuffer hands out Expat's own buffer, fread fills it and XML_ParseBuffer parses it, 64 KiB at a time.

src/native/xml_stream.h
#pragma once
 
#include <expat.h>
 
#include <cstdio>
#include <map>
#include <memory>
#include <stdexcept>
#include <string>
 
// Counts the elements of an XML file by name. The file is read in fixed-size pieces straight into
// Expat's own buffer (XML_GetBuffer, then XML_ParseBuffer), so memory stays the same at any size.
class XmlStream {
public:
// JSON: {"bytes":N,"reads":R,"elements":{"name":count,...}}
static std::string countElements(const std::string& path, int chunkBytes) {
if (chunkBytes < 1024) throw std::invalid_argument("read at least 1 KiB at a time");
std::unique_ptr<FILE, int (*)(FILE*)> file(std::fopen(path.c_str(), "rb"), std::fclose);
if (!file) throw std::runtime_error("cannot open " + path);
std::unique_ptr<XML_ParserStruct, void (*)(XML_Parser)> parser(XML_ParserCreate(nullptr), XML_ParserFree);
if (!parser) throw std::runtime_error("out of memory");
 
std::map<std::string, double> counts;
XML_SetUserData(parser.get(), &counts);
XML_SetStartElementHandler(parser.get(), [](void* data, const XML_Char* name, const XML_Char**) {
(*static_cast<std::map<std::string, double>*>(data))[name] += 1;
});
 
double bytes = 0;
int reads = 0;
for (bool last = false; !last;) {
void* buffer = XML_GetBuffer(parser.get(), chunkBytes);
if (!buffer) throw std::runtime_error("out of memory");
const size_t got = std::fread(buffer, 1, static_cast<size_t>(chunkBytes), file.get());
if (std::ferror(file.get())) throw std::runtime_error("cannot read " + path);
last = got == 0;
bytes += static_cast<double>(got);
reads += last ? 0 : 1;
if (XML_ParseBuffer(parser.get(), static_cast<int>(got), last) != XML_STATUS_OK) {
throw std::runtime_error(std::string(XML_ErrorString(XML_GetErrorCode(parser.get()))) + " at line " +
std::to_string(XML_GetCurrentLineNumber(parser.get())) + ", column " +
std::to_string(XML_GetCurrentColumnNumber(parser.get())));
}
}
 
std::string elements;
for (const auto& [name, count] : counts) elements += (elements.empty() ? "\"" : ",\"") + name + "\":" + std::to_string(static_cast<long long>(count));
return "{\"bytes\":" + std::to_string(static_cast<long long>(bytes)) + ",\"reads\":" + std::to_string(reads) + ",\"elements\":{" + elements + "}}";
}
};
main.js
import { initNative } from './native/xml_stream.h';
 
const m = await initNative();
const { XmlStream } = m;
// A sitemap with 50,000 URLs, the most the sitemap protocol allows in one file.
const urls = Array.from({ length: 50000 }, (_, i) => ` <url><loc>https://example.com/products/${i + 1}</loc><lastmod>2026-09-${String((i % 28) + 1).padStart(2, '0')}</lastmod></url>`);
// m.FS.writeFile adds to a file that already exists, so every run gets a fresh directory.
const dir = await m.getRandomPath('/memfs');
await m.FS.writeFile(`${dir}/sitemap.xml`, `<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n${urls.join('\n')}\n</urlset>\n`);
 
const stats = JSON.parse(await XmlStream.countElements(`${dir}/sitemap.xml`, 65536));
console.log(`${stats.bytes} B in ${stats.reads} reads of 64 KiB`);
for (const [name, count] of Object.entries(stats.elements)) console.log(name, count);
PRINTSfirst run downloads 0.6 MB
4389004 B in 67 reads of 64 KiB
lastmod 50000
loc 50000
url 50000
urlset 1

Read namespaced XML whatever the prefixes

XML_ParserCreateNS resolves every prefix to its namespace URI before your handlers see a name, and XML_SetStartNamespaceDeclHandler reports the declarations themselves.

src/native/xml_names.h
#pragma once
 
#include <expat.h>
 
#include <map>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// Namespace-aware parsing. XML_ParserCreateNS hands every name over as "<namespace URI>|<local name>",
// so what a document calls its prefixes no longer matters; names outside any namespace stay bare.
class XmlNames {
public:
// The text inside every element named `local` in the namespace `uri`, as a JSON array of strings.
static std::string textOf(const std::string& xml, const std::string& uri, const std::string& local) {
Search search;
search.name = uri.empty() ? local : uri + "|" + local;
Parser parser = create(&search);
XML_SetElementHandler(parser.get(), onStart, onEnd);
XML_SetCharacterDataHandler(parser.get(), onText);
parse(parser.get(), xml);
std::string json = "[";
for (const std::string& text : search.found) json += (json.size() > 1 ? "," : "") + quote(text);
return json + "]";
}
 
// Every prefix the document declares, with its URI, as JSON; "" is the default namespace.
static std::string declarations(const std::string& xml) {
std::map<std::string, std::string> declared;
Parser parser = create(&declared);
XML_SetStartNamespaceDeclHandler(parser.get(), [](void* data, const XML_Char* prefix, const XML_Char* uri) {
(*static_cast<std::map<std::string, std::string>*>(data))[prefix ? prefix : ""] = uri ? uri : "";
});
parse(parser.get(), xml);
std::string json = "{";
for (const auto& [prefix, uri] : declared) json += (json.size() > 1 ? "," : "") + quote(prefix) + ":" + quote(uri);
return json + "}";
}
 
private:
using Parser = std::unique_ptr<XML_ParserStruct, void (*)(XML_Parser)>;
 
struct Search {
std::string name;
std::vector<std::string> open; // the text of each open element with that name
std::vector<std::string> found;
};
 
static Parser create(void* data) {
Parser parser(XML_ParserCreateNS(nullptr, '|'), XML_ParserFree);
if (!parser) throw std::runtime_error("out of memory");
XML_SetUserData(parser.get(), data);
return parser;
}
 
static void parse(XML_Parser parser, const std::string& xml) {
if (XML_Parse(parser, xml.data(), static_cast<int>(xml.size()), XML_TRUE) != XML_STATUS_OK) {
throw std::runtime_error(std::string(XML_ErrorString(XML_GetErrorCode(parser))) + " at line " +
std::to_string(XML_GetCurrentLineNumber(parser)) + ", column " +
std::to_string(XML_GetCurrentColumnNumber(parser)));
}
}
 
static void XMLCALL onStart(void* data, const XML_Char* name, const XML_Char**) {
Search& search = *static_cast<Search*>(data);
if (search.name == name) search.open.emplace_back();
}
 
static void XMLCALL onText(void* data, const XML_Char* text, int length) {
Search& search = *static_cast<Search*>(data);
if (!search.open.empty()) search.open.back().append(text, static_cast<size_t>(length));
}
 
static void XMLCALL onEnd(void* data, const XML_Char* name) {
Search& search = *static_cast<Search*>(data);
if (search.name != name) return;
search.found.push_back(search.open.back());
search.open.pop_back();
}
 
static std::string quote(const std::string& value) {
std::string out = "\"";
for (const char c : value) {
if (c == '"' || c == '\\') {
out += '\\';
out += c;
} else if (c == '\n') {
out += "\\n";
} else if (c == '\t') {
out += "\\t";
} else if (c == '\r') {
out += "\\r";
} else {
out += c;
}
}
return out + "\"";
}
};
main.js
import { initNative, XmlNames } from './native/xml_names.h';
 
await initNative();
const GPX = 'http://www.topografix.com/GPX/1/1';
const HR = 'http://www.garmin.com/xmlschemas/TrackPointExtension/v1';
// The same heart rates, written by two exporters that picked different prefixes for Garmin's extension.
const track = (prefix) => `<gpx version="1.1" creator="example" xmlns="${GPX}" xmlns:${prefix}="${HR}">
<trk><trkseg>
<trkpt lat="46.5190" lon="6.5668"><extensions><${prefix}:TrackPointExtension><${prefix}:hr>128</${prefix}:hr></${prefix}:TrackPointExtension></extensions></trkpt>
<trkpt lat="46.5192" lon="6.5671"><extensions><${prefix}:TrackPointExtension><${prefix}:hr>131</${prefix}:hr></${prefix}:TrackPointExtension></extensions></trkpt>
</trkseg></trk>
</gpx>`;
for (const prefix of ['gpxtpx', 'ns3']) {
const rates = JSON.parse(await XmlNames.textOf(track(prefix), HR, 'hr'));
console.log(`${prefix}:hr`, rates.join(' '));
}
console.log('hr in the GPX namespace:', JSON.parse(await XmlNames.textOf(track('gpxtpx'), GPX, 'hr')).length);
console.log(await XmlNames.declarations(track('ns3')));
PRINTSfirst run downloads 0.6 MB
gpxtpx:hr 128 131
ns3:hr 128 131
hr in the GPX namespace: 0
{"":"http://www.topografix.com/GPX/1/1","ns3":"http://www.garmin.com/xmlschemas/TrackPointExtension/v1"}

Expand entities without a billion laughs

Expat expands the entities a document declares and stops the ones that blow up: XML_SetBillionLaughsAttackProtectionMaximumAmplification and …ActivationThreshold set how far.

src/native/xml_guard.h
#pragma once
 
// expat.h declares the amplification limits only when XML_GE is 1; this build compiles them in.
#ifndef XML_GE
#define XML_GE 1
#endif
#include <expat.h>
 
#include <memory>
#include <stdexcept>
#include <string>
 
// Untrusted XML with entities expanded under Expat's limits. Once `activationBytes` of input and
// entity text have been processed, the expansion may be at most `maxAmplification` times the input;
// Expat's defaults are 100 and 8 MiB. A breach stops the parse like any other error.
class XmlGuard {
public:
// The document's character data, entities expanded.
static std::string text(const std::string& xml, double maxAmplification, double activationBytes) {
std::unique_ptr<XML_ParserStruct, void (*)(XML_Parser)> parser(XML_ParserCreate(nullptr), XML_ParserFree);
if (!parser) throw std::runtime_error("out of memory");
if (!XML_SetBillionLaughsAttackProtectionMaximumAmplification(parser.get(), static_cast<float>(maxAmplification))) {
throw std::invalid_argument("the maximum amplification must be at least 1");
}
if (activationBytes < 0 || !XML_SetBillionLaughsAttackProtectionActivationThreshold(parser.get(), static_cast<unsigned long long>(activationBytes))) {
throw std::invalid_argument("the activation threshold must be a byte count");
}
std::string text;
XML_SetUserData(parser.get(), &text);
XML_SetCharacterDataHandler(parser.get(), [](void* data, const XML_Char* chunk, int length) {
static_cast<std::string*>(data)->append(chunk, static_cast<size_t>(length));
});
if (XML_Parse(parser.get(), xml.data(), static_cast<int>(xml.size()), XML_TRUE) != XML_STATUS_OK) {
throw std::runtime_error(std::string(XML_ErrorString(XML_GetErrorCode(parser.get()))) + " at line " +
std::to_string(XML_GetCurrentLineNumber(parser.get())) + ", column " +
std::to_string(XML_GetCurrentColumnNumber(parser.get())));
}
return text;
}
};
main.js
import { initNative, XmlGuard } from './native/xml_guard.h';
 
await initNative();
const MAX_AMPLIFICATION = 100; // Expat's defaults
const THRESHOLD = 8 * 1024 * 1024;
console.log(await XmlGuard.text('<!DOCTYPE note [<!ENTITY product "crossbind">]><note>&product; parses &product;</note>', MAX_AMPLIFICATION, THRESHOLD));
 
// Each level repeats the one below ten times: 9 levels would expand to 3 GB.
const laughs = (levels) => {
const entities = ['<!ENTITY lol0 "lol">'];
for (let level = 1; level <= levels; level += 1) entities.push(`<!ENTITY lol${level} "${`&lol${level - 1};`.repeat(10)}">`);
return `<?xml version="1.0"?>\n<!DOCTYPE lolz [\n${entities.join('\n')}\n]>\n<lolz>&lol${levels};</lolz>`;
};
console.log('5 levels:', (await XmlGuard.text(laughs(5), MAX_AMPLIFICATION, THRESHOLD)).length, 'characters');
try {
await XmlGuard.text(laughs(9), MAX_AMPLIFICATION, THRESHOLD);
} catch (error) {
console.log('9 levels:', error.cppMessage ?? error.message);
}
try {
await XmlGuard.text(laughs(5), MAX_AMPLIFICATION, 64 * 1024);
} catch (error) {
console.log('5 levels, 64 KiB threshold:', error.cppMessage ?? error.message);
}
PRINTSfirst run downloads 0.6 MB
crossbind parses crossbind
5 levels: 300000 characters
9 levels: limit on input amplification factor (from DTD and entities) breached at line 14, column 6
5 levels, 64 KiB threshold: limit on input amplification factor (from DTD and entities) breached at line 10, column 6

What is different on WebAssembly

  • In a browser the module runs in a Worker by default (useWorker), so every call returns a promise: await calls and constructors alike.
  • The module has its own filesystem: m.FS writes files, m.getFileBytes reads them back and m.autoMountFiles mounts File objects from an <input type=file>. /memfs lives in memory; /opfs persists across reloads and needs the Worker. See Filesystem.
  • In Node.js, m.FS is the real disk, so use real paths there.
  • Multi-threaded builds (runtime: 'mt') need COOP and COEP headers in production. See Threading.

Other platforms

Facts on this page come from the port manifests in the repository and from what npm served on beta when the site was built. See the Libraries guide for the full consumer flow.

MORE LIBRARIES
cURLGDALGEOSGeoTIFFiconvLERClibjpeg-turbolibTIFFOpenSSLPROJSpatiaLiteSQLiteWebPzlibZstandard
Type to search every guide page and section.
↑↓ navigate↵ openesc close