Base32 / Base58 / Base62 / Base85 Encoder — C++ source
Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
// (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
// text.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the Base Encoder tool, ported
// from cli/base-encoder/base-encoder.go (the authoritative Go twin).
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws (encode always succeeds, decode
// returns std::nullopt for invalid or malformed input, mirroring the TS
// lib's `null` and the Go twin's `errInvalid`).
// - Functionally equivalent to the Go twin: same inputs -> same outputs.
// - Self-contained: std only — no bignum dependency (boost::multiprecision
// is the ecosystem equivalent the Go twin avoids via math/big).
//
// Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
// array, which overflows any fixed-width integer for inputs longer than a
// few bytes. The Go twin leans on math/big; with no stdlib bignum we
// implement the same idea as the Rust sibling — a little-endian base-256
// std::vector<std::uint8_t> and two primitives: divmod_small (peel a base-N
// digit off the little end) and muladd_small (reassemble a number from its
// base-N digits).
//
// String note: std::string is byte-oriented, so decode returns the decoded
// bytes verbatim — exactly Go's `string(data)` semantics (which never fails
// and never mangles). Languages with validated string types (Rust/Python/...)
// decode lossily; a C++ std::string carries the raw bytes.
#include <cassert>
#include <cstdint>
#include <cstring>
#include <optional>
#include <string>
#include <string_view>
#include <vector>
namespace base_encoder {
// Selects a byte-array base encoding. Mirrors the Go twin's `Scheme` type
// (and the TS `Scheme` union 'base32' | 'base58' | 'base62' | 'base85').
enum class Scheme { Base32, Base58, Base62, Base85 };
constexpr std::string_view kAlphabet32 = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
constexpr std::string_view kAlphabet58 =
"123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
constexpr std::string_view kAlphabet62 =
"0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
// Maps the size of a final (partial) 5-byte chunk to the number of data
// characters it emits before '=' padding, per RFC 4648. Index = byte count
// (0..=4). Matches `outLen = [0, 2, 4, 5, 7][chunk.length]` in the TS.
constexpr int kOutLen32[5] = {0, 2, 4, 5, 7};
// ---------------------------------------------------------------------------
// Arbitrary-precision primitives (base-256, little-endian). Used by Base58
// and Base62 so the port stays dependency-free.
// ---------------------------------------------------------------------------
// Divide a little-endian base-256 unsigned integer by a small `base`
// (<= 256), storing the quotient back into `digits` (with high zero limbs
// stripped) and returning the remainder. The long-division step used to
// peel base-N digits off the little end during encoding.
unsigned divmodSmall(std::vector<std::uint8_t>& digits, unsigned base) {
unsigned rem = 0;
for (auto it = digits.rbegin(); it != digits.rend(); ++it) {
unsigned cur = rem * 256 + *it;
*it = static_cast<std::uint8_t>(cur / base);
rem = cur % base;
}
// Strip high (trailing in LE) zero limbs — keeps the representation minimal.
while (!digits.empty() && digits.back() == 0) digits.pop_back();
return rem;
}
// Multiply a little-endian base-256 unsigned integer by `base` and add
// `digit`, in place. The inverse of divmodSmall: reassembles a number from
// its base-N digits (processed most-significant first).
void muladdSmall(std::vector<std::uint8_t>& digits, unsigned base, unsigned digit) {
unsigned carry = digit;
for (auto& d : digits) {
unsigned cur = static_cast<unsigned>(d) * base + carry;
d = static_cast<std::uint8_t>(cur & 0xff);
carry = cur >> 8;
}
while (carry > 0) {
digits.push_back(static_cast<std::uint8_t>(carry & 0xff));
carry >>= 8;
}
}
// Little-endian base-256 -> minimal big-endian bytes (the form the encoders
// emit and the decoders reconstruct). Strips any accidental leading zero so
// the output matches Go's `big.Int.Bytes()` exactly.
std::vector<std::uint8_t> toBeBytes(std::vector<std::uint8_t> digits) {
std::reverse(digits.begin(), digits.end());
while (!digits.empty() && digits.front() == 0) {
digits.erase(digits.begin());
}
return digits;
}
// Looks up a character in an alphabet; -1 when absent (the TS/Python `.find`
// / `.indexOf` returning -1 or nil).
int indexOf(std::string_view alphabet, char c) {
auto pos = alphabet.find(c);
return pos == std::string_view::npos ? -1 : static_cast<int>(pos);
}
// ---------------------------------------------------------------------------
// Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
// ---------------------------------------------------------------------------
std::string encode32(std::string_view data) {
std::string out;
for (std::size_t i = 0; i < data.size(); i += 5) {
std::size_t n = std::min<std::size_t>(5, data.size() - i);
unsigned b[5] = {0, 0, 0, 0, 0};
for (std::size_t j = 0; j < n; j++) {
b[j] = static_cast<unsigned char>(data[i + j]);
}
// Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
unsigned digits[8] = {
(b[0] >> 3) & 0x1f,
((b[0] << 2) | (b[1] >> 6)) & 0x1f,
(b[1] >> 1) & 0x1f,
((b[1] << 4) | (b[2] >> 4)) & 0x1f,
((b[2] << 1) | (b[3] >> 7)) & 0x1f,
(b[3] >> 2) & 0x1f,
((b[3] << 3) | (b[4] >> 5)) & 0x1f,
b[4] & 0x1f,
};
int outLen = n == 5 ? 8 : kOutLen32[n];
for (int k = 0; k < outLen; k++) {
out.push_back(kAlphabet32[digits[k]]);
}
for (int k = outLen; k < 8; k++) {
out.push_back('=');
}
}
return out;
}
std::optional<std::vector<std::uint8_t>> decode32(std::string_view s) {
std::vector<std::uint8_t> out;
unsigned buffer = 0;
int bits = 0;
for (char c : s) {
if (c == '=') {
break; // padding marks the end
}
int idx = indexOf(kAlphabet32, c);
if (idx < 0) {
return std::nullopt;
}
buffer = (buffer << 5) | static_cast<unsigned>(idx);
bits += 5;
if (bits >= 8) {
bits -= 8;
out.push_back(static_cast<std::uint8_t>((buffer >> bits) & 0xff));
buffer &= (1u << bits) - 1u; // keep only the leftover bits
}
}
return out;
}
// ---------------------------------------------------------------------------
// Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
// preserved).
// ---------------------------------------------------------------------------
std::string encode58(std::string_view data) {
// Count leading zero bytes — each maps to a leading '1'.
std::size_t zeros = 0;
while (zeros < data.size() && data[zeros] == '\0') {
zeros++;
}
// Big-endian byte array (skipping the leading zeros) -> LE base-256.
std::vector<std::uint8_t> le;
for (std::size_t i = zeros; i < data.size(); i++) {
muladdSmall(le, 256, static_cast<unsigned char>(data[i]));
}
// Base-convert to 58 digits (collected least-significant first; every
// digit is < 58, so they fit in single bytes).
std::vector<std::uint8_t> digits;
while (!le.empty()) {
digits.push_back(static_cast<std::uint8_t>(divmodSmall(le, 58)));
}
std::string out(zeros, '1');
for (auto it = digits.rbegin(); it != digits.rend(); ++it) {
out.push_back(kAlphabet58[*it]);
}
return out;
}
std::optional<std::vector<std::uint8_t>> decode58(std::string_view s) {
// Count leading '1's — each maps to a 0x00 byte.
std::size_t zeros = 0;
while (zeros < s.size() && s[zeros] == '1') {
zeros++;
}
std::vector<std::uint8_t> le;
for (std::size_t i = zeros; i < s.size(); i++) {
int idx = indexOf(kAlphabet58, s[i]);
if (idx < 0) {
return std::nullopt;
}
muladdSmall(le, 58, static_cast<unsigned>(idx));
}
// LE -> minimal big-endian bytes.
std::vector<std::uint8_t> out(zeros, 0);
auto body = toBeBytes(std::move(le));
out.insert(out.end(), body.begin(), body.end());
return out;
}
// ---------------------------------------------------------------------------
// Base62 — standard base-conversion of the byte array (no leading-zero
// special-casing beyond the standard big-int).
// ---------------------------------------------------------------------------
std::string encode62(std::string_view data) {
if (data.empty()) {
return std::string();
}
std::vector<std::uint8_t> le;
for (char c : data) {
muladdSmall(le, 256, static_cast<unsigned char>(c));
}
if (le.empty()) {
return "0"; // value zero
}
std::vector<std::uint8_t> digits;
while (!le.empty()) {
digits.push_back(static_cast<std::uint8_t>(divmodSmall(le, 62)));
}
std::string out;
for (auto it = digits.rbegin(); it != digits.rend(); ++it) {
out.push_back(kAlphabet62[*it]);
}
return out;
}
std::optional<std::vector<std::uint8_t>> decode62(std::string_view s) {
if (s.empty()) {
return std::vector<std::uint8_t>{};
}
std::vector<std::uint8_t> le;
for (char c : s) {
int idx = indexOf(kAlphabet62, c);
if (idx < 0) {
return std::nullopt;
}
muladdSmall(le, 62, static_cast<unsigned>(idx));
}
return toBeBytes(std::move(le));
}
// ---------------------------------------------------------------------------
// Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
// group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
// one fewer char than (bytes+1) would suggest; decode reverses, padding with
// 'u' (value 84).
// ---------------------------------------------------------------------------
std::string encode85(std::string_view data) {
std::string out;
for (std::size_t i = 0; i < data.size(); i += 4) {
std::size_t n = std::min<std::size_t>(4, data.size() - i);
bool isFull = n == 4;
unsigned b[4] = {0, 0, 0, 0};
for (std::size_t j = 0; j < n; j++) {
b[j] = static_cast<unsigned char>(data[i + j]);
}
unsigned u = b[0] * 16777216u + b[1] * 65536u + b[2] * 256u + b[3];
if (isFull && u == 0) {
out.push_back('z'); // zero-group shorthand
continue;
}
unsigned digits[5] = {0, 0, 0, 0, 0};
unsigned v = u;
for (int k = 4; k >= 0; k--) {
digits[k] = v % 85;
v /= 85;
}
std::size_t emit = isFull ? 5 : n + 1; // n bytes -> n+1 chars
for (std::size_t k = 0; k < emit; k++) {
out.push_back(static_cast<char>(digits[k] + 33));
}
}
return out;
}
std::optional<std::vector<std::uint8_t>> decode85(std::string_view s) {
std::vector<std::uint8_t> out;
std::vector<unsigned> group; // accumulated digit values (0..84)
group.reserve(5);
for (char ch : s) {
unsigned c = static_cast<unsigned char>(ch);
if (c == 'z') {
// 'z' is only valid at a group boundary (an empty accumulator).
if (!group.empty()) {
return std::nullopt;
}
out.insert(out.end(), {0, 0, 0, 0});
continue;
}
if (c < 33 || c > 117) {
return std::nullopt;
}
group.push_back(c - 33);
if (group.size() == 5) {
unsigned long long v = 0;
for (unsigned d : group) {
v = v * 85 + d;
}
if (v > 0xFFFFFFFFull) {
return std::nullopt; // a 5-char group must fit in 32 bits
}
out.push_back(static_cast<std::uint8_t>((v >> 24) & 0xff));
out.push_back(static_cast<std::uint8_t>((v >> 16) & 0xff));
out.push_back(static_cast<std::uint8_t>((v >> 8) & 0xff));
out.push_back(static_cast<std::uint8_t>(v & 0xff));
group.clear();
}
}
// Handle a partial final group (2-4 chars -> 1-3 bytes).
if (!group.empty()) {
std::size_t m = group.size();
if (m < 2) {
return std::nullopt; // a lone trailing char is malformed
}
while (group.size() < 5) {
group.push_back(84); // pad with 'u'
}
unsigned long long v = 0;
for (unsigned d : group) {
v = v * 85 + d;
}
if (v > 0xFFFFFFFFull) {
return std::nullopt;
}
const std::uint8_t all[4] = {
static_cast<std::uint8_t>((v >> 24) & 0xff),
static_cast<std::uint8_t>((v >> 16) & 0xff),
static_cast<std::uint8_t>((v >> 8) & 0xff),
static_cast<std::uint8_t>(v & 0xff),
};
out.insert(out.end(), all, all + (m - 1));
}
return out;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
// Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
// private `encodeBytes`.
std::string encodeBytes(std::string_view data, Scheme scheme) {
switch (scheme) {
case Scheme::Base32: return encode32(data);
case Scheme::Base58: return encode58(data);
case Scheme::Base62: return encode62(data);
case Scheme::Base85: return encode85(data);
}
return std::string();
}
// Dispatch an encoded string to the chosen scheme's decoder. An invalid or
// malformed input yields std::nullopt (mirroring the TS `null`). Mirrors the
// Go twin's private `decodeBytes`.
std::optional<std::vector<std::uint8_t>> decodeBytes(std::string_view encoded, Scheme scheme) {
switch (scheme) {
case Scheme::Base32: return decode32(encoded);
case Scheme::Base58: return decode58(encoded);
case Scheme::Base62: return decode62(encoded);
case Scheme::Base85: return decode85(encoded);
}
return std::nullopt;
}
// Returns the chosen-scheme encoding of the UTF-8 bytes of `text`. Empty text
// encodes to "". It is the C++ twin of `Encode` in
// cli/base-encoder/base-encoder.go.
std::string encode(std::string_view text, Scheme scheme) {
return encodeBytes(text, scheme);
}
// Reverses an encoded string back to the decoded bytes (held verbatim in a
// std::string — Go's `string(data)`). Invalid characters or a malformed
// structure yield std::nullopt — mirroring the Go twin's `errInvalid` and
// the TS lib's `null`. It is the C++ twin of `Decode` in
// cli/base-encoder/base-encoder.go.
std::optional<std::string> decode(std::string_view encoded, Scheme scheme) {
auto bytes = decodeBytes(encoded, scheme);
if (!bytes) {
return std::nullopt;
}
return std::string(bytes->begin(), bytes->end());
}
} // namespace base_encoder
// ---------------------------------------------------------------------------
// Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
// Run directly: `c++ -std=c++17 cpp.cpp && ./a.out`
// ---------------------------------------------------------------------------
int main() {
using base_encoder::Scheme;
// Base32 — known values + RFC 4648 padding + case sensitivity.
assert(base_encoder::encode("hello", Scheme::Base32) == "NBSWY3DP");
// 3 bytes -> 5 data chars + 3 '=' pads.
assert(base_encoder::encode("foo", Scheme::Base32) == "MZXW6===");
assert(base_encoder::decode("NBSWY3DP", Scheme::Base32) == std::optional<std::string>("hello"));
// lowercase is not in the RFC 4648 alphabet
assert(!base_encoder::decode("nbswy3dp", Scheme::Base32).has_value());
// Base58 — each leading 0x00 byte -> a leading '1'.
assert(base_encoder::encode(std::string_view("\0", 1), Scheme::Base58) == "1");
assert(base_encoder::encode(std::string_view("\0\0A", 3), Scheme::Base58)
.rfind("11", 0) == 0);
assert(base_encoder::decode("1", Scheme::Base58) == std::optional<std::string>(std::string_view("\0", 1)));
// round-trip preserves the leading zero bytes exactly
std::string zerosA(std::string_view("\0\0A", 3));
assert(base_encoder::decode(base_encoder::encode(zerosA, Scheme::Base58), Scheme::Base58) ==
std::optional<std::string>(zerosA));
// Base62 — plain big-int base conversion (no leading-zero preservation).
assert(base_encoder::encode("A", Scheme::Base62) == "13"); // 1*62 + 3
assert(base_encoder::decode("13", Scheme::Base62) == std::optional<std::string>("A"));
assert(base_encoder::encode(std::string_view("\0", 1), Scheme::Base62) == "0");
// no leading-zero preservation: the minimal rep of 0 is empty
assert(base_encoder::decode("0", Scheme::Base62) == std::optional<std::string>(""));
// Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection.
assert(base_encoder::encode("hello", Scheme::Base85) == "BOu!rDZ");
assert(base_encoder::encode(std::string_view("\0\0\0\0", 4), Scheme::Base85) == "z");
assert(base_encoder::encode(std::string_view("\0\0\0\0\0\0\0\0", 8), Scheme::Base85) == "zz");
// a 5-char group must fit in 32 bits; "uuuuu" overflows
assert(!base_encoder::decode("uuuuu", Scheme::Base85).has_value());
// a lone trailing char is a malformed partial group
assert(!base_encoder::decode("B", Scheme::Base85).has_value());
// Cross-scheme — empty, multibyte round-trip, and invalid rejection.
const Scheme schemes[] = {Scheme::Base32, Scheme::Base58, Scheme::Base62, Scheme::Base85};
for (Scheme scheme : schemes) {
assert(base_encoder::encode("", scheme).empty());
assert(base_encoder::decode("", scheme) == std::optional<std::string>(""));
// multibyte UTF-8 round-trips through every scheme
assert(base_encoder::decode(base_encoder::encode("CosmoDev 🚀", scheme), scheme) ==
std::optional<std::string>("CosmoDev 🚀"));
// '~' is outside every supported alphabet
assert(!base_encoder::decode("~!not-valid!~", scheme).has_value());
}
puts("ok");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →