crossbind
GitHub

iconv for Android

v1.19Android

iconv 1.19 for React Native apps on Android, precompiled for arm64-v8a devices and the x86_64 emulator as @crossbind/port-iconv-android.

npm install @crossbind/port-iconv-android@beta

Install

shell
npm install @crossbind/plugin-react-native@beta @crossbind/plugin-react-native-ios-helper@beta @crossbind/port-iconv-android@beta
npm install --save-dev @crossbind/plugin-metro@beta
crossbind.config.mjs
import iconvAndroid from '@crossbind/port-iconv-android/crossbind.config.js';
 
export default {
dependencies: [iconvAndroid],
paths: { config: import.meta.url },
};
metro.config.js
const { getDefaultConfig, mergeConfig } = require('@react-native/metro-config');
const CrossbindMetroPlugin = require('@crossbind/plugin-metro');
 
const defaultConfig = getDefaultConfig(__dirname);
 
const config = {
...CrossbindMetroPlugin(defaultConfig),
};
 
module.exports = mergeConfig(defaultConfig, config);

The whole flow, including Expo, is in the React Native playbook.

Usage

The examples the WebAssembly page runs, as Android compiles them: the same headers and the same calls. They are checked on the WebAssembly build.

Convert between UTF-8 and a legacy encoding

What most iconv code does: iconv_open, iconv in a loop and iconv_close, strict in both directions. A last call without input ends stateful encodings such as ISO-2022-JP, which is where the closing ESC ( B comes from.

src/native/charset.h
#pragma once
 
#include <iconv.h>
 
#include <cerrno>
#include <cstdio>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// Strict conversion between UTF-8 text and the bytes of any encoding libiconv knows. Bytes cross the
// binding as a byte string: one UTF-16 code unit (0-255) per byte.
class Charset {
public:
static std::u16string encode(const std::string& text, const std::string& encoding) {
const std::string bytes = convert(text, "UTF-8", encoding);
std::u16string units(bytes.size(), u'\0');
for (size_t i = 0; i < bytes.size(); ++i) units[i] = static_cast<unsigned char>(bytes[i]);
return units;
}
 
static std::string decode(const std::u16string& bytes, const std::string& encoding) {
std::string input(bytes.size(), '\0');
for (size_t i = 0; i < bytes.size(); ++i) {
if (bytes[i] > 0xFF) throw std::invalid_argument("not a byte string");
input[i] = static_cast<char>(bytes[i]);
}
return convert(input, encoding, "UTF-8");
}
 
private:
static std::string convert(const std::string& input, const std::string& from, const std::string& to) {
const iconv_t opened = iconv_open(to.c_str(), from.c_str());
if (opened == reinterpret_cast<iconv_t>(-1)) throw std::invalid_argument("iconv cannot convert " + from + " to " + to);
std::unique_ptr<void, int (*)(iconv_t)> cd(opened, iconv_close);
std::string output;
std::vector<char> buffer(4096);
char* in = const_cast<char*>(input.data());
size_t inLeft = input.size();
for (;;) {
char* out = buffer.data();
size_t outLeft = buffer.size();
// Once the input is used up, a call without input ends a stateful encoding (ISO-2022-JP
// switches back to ASCII); for the others it writes nothing.
const bool finishing = inLeft == 0;
const size_t status = finishing ? iconv(cd.get(), nullptr, nullptr, &out, &outLeft) : iconv(cd.get(), &in, &inLeft, &out, &outLeft);
output.append(buffer.data(), buffer.size() - outLeft);
if (status != static_cast<size_t>(-1)) {
if (finishing) return output;
} else if (errno != E2BIG) { // E2BIG only means the buffer is full, and it was just emptied
const size_t at = input.size() - inLeft;
if (errno == EINVAL) throw std::runtime_error("incomplete " + from + " input at byte " + std::to_string(at));
if (from != "UTF-8") throw std::runtime_error("invalid " + from + " input at byte " + std::to_string(at));
throw std::runtime_error(missing(input, at, to));
}
}
}
 
// Text coming from JavaScript is valid UTF-8, so EILSEQ means the target has no such character.
static std::string missing(const std::string& text, size_t at, const std::string& to) {
const auto lead = static_cast<unsigned char>(text[at]);
const size_t length = lead < 0x80 ? 1 : lead < 0xE0 ? 2 : lead < 0xF0 ? 3 : 4;
unsigned long codePoint = length == 1 ? lead : lead & (0xFF >> (length + 1));
for (size_t i = 1; i < length; ++i) codePoint = (codePoint << 6) | (static_cast<unsigned char>(text[at + i]) & 0x3F);
size_t character = 0;
for (size_t i = 0; i < at; ++i) character += (static_cast<unsigned char>(text[i]) & 0xC0) != 0x80;
char name[16];
std::snprintf(name, sizeof name, "U+%04lX", codePoint);
return "cannot encode " + std::string(name) + " at character " + std::to_string(character) + " in " + to;
}
};
main.js
import { initNative, Charset } from './native/charset.h';
 
await initNative();
const hex = (bytes) => [...bytes].map((c) => c.charCodeAt(0).toString(16).padStart(2, '0')).join(' ');
const sjis = await Charset.encode('日本語', 'SHIFT_JIS');
console.log(hex(sjis));
console.log(await Charset.decode(sjis, 'SHIFT_JIS'));
console.log(hex(await Charset.encode('日本語', 'ISO-2022-JP')));
try {
await Charset.encode('Total: €5', 'ISO-8859-1');
} catch (error) {
console.log(error.cppMessage ?? error.message);
}
PRINTS
93 fa 96 7b 8c ea
日本語
1b 24 42 46 7c 4b 5c 38 6c 1b 28 42
cannot encode U+20AC at character 7 in ISO-8859-1

Transliterate or drop what the target cannot hold

Append //TRANSLIT to the target and iconv approximates a missing character; append //IGNORE and it drops it. The return value of iconv counts those characters. GNU libiconv transliterates the same way in every locale, so ASCII keeps accents as marks.

src/native/legacy_encoder.h
#pragma once
 
#include <iconv.h>
 
#include <cerrno>
#include <memory>
#include <stdexcept>
#include <string>
 
// Encodes UTF-8 text for a target named the way iconv_open takes it: "ISO-8859-1" stops at the first
// character the encoding lacks, "ISO-8859-1//TRANSLIT" approximates it (€ becomes EUR) and
// "ISO-8859-1//IGNORE" drops it. The descriptor is opened once and reused for every call.
class LegacyEncoder {
public:
explicit LegacyEncoder(const std::string& target) : cd(open(target), iconv_close), target(target) {}
 
std::u16string encode(const std::string& text) {
// iconv returns how many characters it changed only when a call succeeds, so the whole text
// goes in one call, into a buffer that grows until it fits.
for (size_t capacity = text.size() * 2 + 16;; capacity *= 2) {
iconv(cd.get(), nullptr, nullptr, nullptr, nullptr); // back to the initial state
std::string output(capacity, '\0');
char* in = const_cast<char*>(text.data());
size_t inLeft = text.size();
char* out = &output[0];
size_t outLeft = capacity;
const size_t status = iconv(cd.get(), &in, &inLeft, &out, &outLeft);
if (status == static_cast<size_t>(-1) && errno != E2BIG) throw std::runtime_error(missing(text, text.size() - inLeft));
if (status == static_cast<size_t>(-1) || iconv(cd.get(), nullptr, nullptr, &out, &outLeft) == static_cast<size_t>(-1)) continue;
changes = status;
std::u16string bytes(capacity - outLeft, u'\0');
for (size_t i = 0; i < bytes.size(); ++i) bytes[i] = static_cast<unsigned char>(output[i]);
return bytes;
}
}
 
// How many characters the last encode approximated or dropped.
int changed() const { return static_cast<int>(changes); }
 
private:
static iconv_t open(const std::string& target) {
const iconv_t cd = iconv_open(target.c_str(), "UTF-8");
if (cd == reinterpret_cast<iconv_t>(-1)) throw std::invalid_argument("iconv cannot encode to " + target);
return cd;
}
 
std::string missing(const std::string& text, size_t at) const {
const auto lead = static_cast<unsigned char>(text[at]);
const size_t length = lead < 0x80 ? 1 : lead < 0xE0 ? 2 : lead < 0xF0 ? 3 : 4;
size_t character = 0;
for (size_t i = 0; i < at; ++i) character += (static_cast<unsigned char>(text[i]) & 0xC0) != 0x80;
return target + " has no \"" + text.substr(at, length) + "\" (character " + std::to_string(character) + ")";
}
 
std::unique_ptr<void, int (*)(iconv_t)> cd;
std::string target;
size_t changes = 0;
};
main.js
import { initNative, LegacyEncoder } from './native/legacy_encoder.h';
import { Charset } from './native/charset.h';
 
await initNative();
const text = 'Crème brûlée – 5 € “spécial”';
for (const target of ['ISO-8859-1//TRANSLIT', 'ISO-8859-1//IGNORE', 'ASCII//TRANSLIT']) {
const encoder = await new LegacyEncoder(target);
const bytes = await encoder.encode(text);
console.log(`${target}: ${await Charset.decode(bytes, 'ISO-8859-1')} (${await encoder.changed()} changed)`);
}
PRINTS
ISO-8859-1//TRANSLIT: Crème brûlée - 5 EUR "spécial" (4 changed)
ISO-8859-1//IGNORE: Crème brûlée  5  spécial (4 changed)
ASCII//TRANSLIT: Cr`eme br^ul'ee - 5 EUR "sp'ecial" (8 changed)

Decode a stream that cuts characters in two

Bytes from fetch, a WebSocket or a serial port arrive in pieces of any size, so a multibyte character can straddle two pieces. iconv stops with EINVAL at such a cut and the unfinished bytes wait for the next piece; E2BIG only means the output buffer is full.

src/native/stream_decoder.h
#pragma once
 
#include <iconv.h>
 
#include <cerrno>
#include <memory>
#include <stdexcept>
#include <string>
#include <vector>
 
// Decodes bytes that arrive in pieces (a fetch body, a WebSocket, a serial port) into UTF-8 text.
// A piece can end in the middle of a multibyte character: iconv then stops with EINVAL, and those
// bytes wait for the next piece. The calls follow Node's StringDecoder: write() every piece, then end().
class StreamDecoder {
public:
explicit StreamDecoder(const std::string& encoding) : cd(open(encoding), iconv_close), encoding(encoding) {}
 
std::string write(const std::u16string& piece) {
for (char16_t unit : piece) {
if (unit > 0xFF) throw std::invalid_argument("not a byte string");
pending += static_cast<char>(unit);
}
std::string text;
std::vector<char> buffer(4096);
char* in = &pending[0];
size_t inLeft = pending.size();
while (inLeft > 0) {
char* out = buffer.data();
size_t outLeft = buffer.size();
const size_t status = iconv(cd.get(), &in, &inLeft, &out, &outLeft);
text.append(buffer.data(), buffer.size() - outLeft);
if (status != static_cast<size_t>(-1) || errno == E2BIG) continue;
if (errno == EINVAL) { // the piece ends inside a character: keep its bytes for the next one
cuts += 1;
break;
}
throw std::runtime_error("invalid " + encoding + " input at byte " + std::to_string(position + pending.size() - inLeft));
}
position += pending.size() - inLeft;
pending.erase(0, pending.size() - inLeft);
return text;
}
 
// The stream is over: bytes still waiting are a character that never finished.
void end() {
const size_t left = pending.size();
pending.clear();
position = 0;
iconv(cd.get(), nullptr, nullptr, nullptr, nullptr);
if (left) throw std::runtime_error(encoding + " stream ends inside a character, " + std::to_string(left) + " byte(s) left over");
}
 
// How many pieces ended inside a character.
int splits() const { return cuts; }
 
private:
static iconv_t open(const std::string& encoding) {
const iconv_t cd = iconv_open("UTF-8", encoding.c_str());
if (cd == reinterpret_cast<iconv_t>(-1)) throw std::invalid_argument("iconv cannot decode " + encoding);
return cd;
}
 
std::unique_ptr<void, int (*)(iconv_t)> cd;
std::string encoding;
std::string pending;
size_t position = 0;
int cuts = 0;
};
main.js
import { initNative, StreamDecoder } from './native/stream_decoder.h';
import { Charset } from './native/charset.h';
 
await initNative();
const bytes = await Charset.encode('配送先: 東京都渋谷区神南1-2-3', 'SHIFT_JIS');
const decoder = await new StreamDecoder('SHIFT_JIS');
let text = '';
for (let at = 0; at < bytes.length; at += 3) text += await decoder.write(bytes.slice(at, at + 3));
await decoder.end();
console.log(`${bytes.length} bytes in pieces of 3, ${await decoder.splits()} characters cut in two`);
console.log(text);
 
const cut = await new StreamDecoder('SHIFT_JIS');
await cut.write(bytes.slice(0, 3));
try {
await cut.end();
} catch (error) {
console.log(error.cppMessage ?? error.message);
}
PRINTS
29 bytes in pieces of 3, 4 characters cut in two
配送先: 東京都渋谷区神南1-2-3
SHIFT_JIS stream ends inside a character, 1 byte(s) left over

List the encodings and check a name

iconvlist walks every encoding with its aliases and iconv_canonicalize names the canonical one, but only iconv_open proves a name works. GNU libiconv writes UTF-8 with its hyphen, and this build leaves out CP437 with the other DOS and EBCDIC code pages.

src/native/encodings.h
#pragma once
 
#include <iconv.h>
 
#include <string>
 
// What this build of libiconv converts. iconvlist() walks every encoding with all of its names and
// iconv_canonicalize() maps an alias to the canonical one. iconv_open() is the real test: an unknown
// name comes back from iconv_canonicalize unchanged.
class Encodings {
public:
// "1.19": _libiconv_version holds (major << 8) + minor.
static std::string version() { return std::to_string(_libiconv_version >> 8) + "." + std::to_string(_libiconv_version & 0xFF); }
 
// One entry per encoding, every name it answers to: [["CP1252","MS-ANSI","WINDOWS-1252"], ...].
static std::string list() {
std::string json = "[";
iconvlist(addEncoding, &json);
return json + "]";
}
 
// The canonical name, or "" when this build cannot convert from or to that encoding.
static std::string resolve(const std::string& name) {
const iconv_t cd = iconv_open("UTF-8", name.c_str());
if (cd == reinterpret_cast<iconv_t>(-1)) return "";
iconv_close(cd);
return iconv_canonicalize(name.c_str());
}
 
private:
static int addEncoding(unsigned int count, const char* const* names, void* data) {
std::string& json = *static_cast<std::string*>(data);
json += json.size() > 1 ? ",[" : "[";
for (unsigned int i = 0; i < count; ++i) json += std::string(i ? ",\"" : "\"") + names[i] + "\"";
json += "]";
return 0; // non-zero would stop the walk
}
};
main.js
import { initNative, Encodings } from './native/encodings.h';
 
await initNative();
const encodings = JSON.parse(await Encodings.list());
console.log(`libiconv ${await Encodings.version()}: ${encodings.length} encodings under ${encodings.flat().length} names`);
for (const name of ['latin1', 'SJIS', 'windows-1254', 'UTF8', 'CP437']) {
console.log(`${name}: ${(await Encodings.resolve(name)) || 'not supported'}`);
}
PRINTS
libiconv 1.19: 112 encodings under 349 names
latin1: ISO-8859-1
SJIS: SHIFT_JIS
windows-1254: CP1254
UTF8: not supported
CP437: not supported

What is different on Android

  • The React Native plugin compiles your headers with the library inside Gradle's native build, so npm run android builds everything.
  • Named imports from ./native/<header>.h work as on the web: await initNative() once, then call the classes.
  • There is no m.FS and no /memfs: files live in the app's own storage, and your C++ takes their paths.
  • No Worker and no COOP or COEP: runtime: 'mt' uses pthreads directly.

Other platforms

Facts on this page come from the port manifests in the repository and from what npm served on beta when the site was built. See the Libraries guide for the full consumer flow.

MORE LIBRARIES
cURLExpatGDALGEOSGeoTIFFLERClibjpeg-turbolibTIFFOpenSSLPROJSpatiaLiteSQLiteWebPzlibZstandard
Type to search every guide page and section.
↑↓ navigate↵ openesc close