Base32 / Base58 / Base62 / Base85 Encoder — C source
Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
* (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
* text.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the Base Encoder tool, ported
* from cli/base-encoder/base-encoder.go (the authoritative Go twin).
* License: display source — part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; never crashes (encode always succeeds, decode
* returns NULL for invalid or malformed input, mirroring the TS lib's
* `null` and the Go twin's `errInvalid`).
* - Functionally equivalent to the Go twin: same inputs -> same outputs.
* - Self-contained: stdlib only — no bignum library (the GMP/tommath
* dependency the Go twin avoids via math/big is not pulled in either).
*
* Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
* array, which overflows any fixed-width integer for inputs longer than a
* few bytes. The Go twin leans on math/big; with no stdlib bignum we
* implement the same idea as the Rust sibling — a little-endian base-256
* byte buffer and two primitives: divmod_small (peel a base-N digit off the
* little end) and muladd_small (reassemble a number from its base-N digits).
*
* String note: C strings are byte arrays, so decode returns the raw bytes —
* exactly Go's `string(data)` semantics (which never fails and never
* mangles). Languages with validated string types (Rust/Python/...) decode
* lossily; here the caller receives the bytes verbatim.
*/
#include <assert.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* One of the four supported byte-array base encodings. Mirrors the Go twin's
* `Scheme` type and the TS `Scheme` union. */
typedef enum { SCHEME_BASE32, SCHEME_BASE58, SCHEME_BASE62, SCHEME_BASE85 } scheme_t;
static const char B32_ALPHABET[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
static const char B58_ALPHABET[] =
"123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
static const char B62_ALPHABET[] =
"0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
/* Data characters emitted by a final (partial) 5-byte chunk before '='
* padding, per RFC 4648. Index = byte count (0..4). Matches the TS `outLen`
* table. */
static const int OUT_LEN_32[5] = {0, 2, 4, 5, 7};
/* ---------------------------------------------------------------------------
* Growable byte buffer — used both for output strings and for the
* little-endian base-256 bignum limbs. Aborts on OOM (showcase simplicity).
* ------------------------------------------------------------------------ */
typedef struct {
unsigned char *data;
size_t len;
size_t cap;
} buf_t;
static void buf_reserve(buf_t *b, size_t extra) {
if (b->len + extra <= b->cap) return;
size_t cap = b->cap ? b->cap : 16;
while (cap < b->len + extra) cap *= 2;
unsigned char *grown = realloc(b->data, cap);
if (!grown) abort();
b->data = grown;
b->cap = cap;
}
static void buf_push(buf_t *b, unsigned char c) {
buf_reserve(b, 1);
b->data[b->len++] = c;
}
/* ---------------------------------------------------------------------------
* Arbitrary-precision primitives (base-256, little-endian). Used by Base58
* and Base62 so the port stays dependency-free.
* ------------------------------------------------------------------------ */
/* Divide a little-endian base-256 unsigned integer by a small `base`
* (<= 256), storing the quotient back in `le` (with high zero limbs
* stripped) and returning the remainder. The long-division step used to
* peel base-N digits off the little end during encoding. */
static unsigned divmod_small(buf_t *le, unsigned base) {
unsigned rem = 0;
for (size_t i = le->len; i > 0; i--) {
unsigned cur = rem * 256 + le->data[i - 1];
le->data[i - 1] = (unsigned char)(cur / base);
rem = cur % base;
}
while (le->len > 0 && le->data[le->len - 1] == 0) le->len--;
return rem;
}
/* Multiply a little-endian base-256 unsigned integer by `base` and add
* `digit`, in place. The inverse of divmod_small: reassembles a number from
* its base-N digits (processed most-significant first). */
static void muladd_small(buf_t *le, unsigned base, unsigned digit) {
unsigned carry = digit;
for (size_t i = 0; i < le->len; i++) {
unsigned cur = (unsigned)le->data[i] * base + carry;
le->data[i] = (unsigned char)(cur & 0xff);
carry = cur >> 8;
}
while (carry > 0) {
buf_push(le, (unsigned char)(carry & 0xff));
carry >>= 8;
}
}
/* ---------------------------------------------------------------------------
* Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
* ------------------------------------------------------------------------ */
static void encode32(buf_t *out, const unsigned char *data, size_t len) {
size_t i = 0;
while (i < len) {
size_t n = len - i < 5 ? len - i : 5;
unsigned b[5] = {0, 0, 0, 0, 0};
for (size_t j = 0; j < n; j++) b[j] = data[i + j];
/* Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian). */
unsigned digits[8] = {
(b[0] >> 3) & 0x1f,
((b[0] << 2) | (b[1] >> 6)) & 0x1f,
(b[1] >> 1) & 0x1f,
((b[1] << 4) | (b[2] >> 4)) & 0x1f,
((b[2] << 1) | (b[3] >> 7)) & 0x1f,
(b[3] >> 2) & 0x1f,
((b[3] << 3) | (b[4] >> 5)) & 0x1f,
b[4] & 0x1f,
};
size_t out_len = n == 5 ? 8 : (size_t)OUT_LEN_32[n];
for (size_t k = 0; k < out_len; k++) buf_push(out, (unsigned char)B32_ALPHABET[digits[k]]);
for (size_t k = out_len; k < 8; k++) buf_push(out, '=');
i += 5;
}
}
static unsigned char *decode32(const char *s, size_t *out_len) {
buf_t out = {0};
buf_reserve(&out, 1); /* keep data non-NULL for the empty case */
unsigned buffer = 0, bits = 0;
for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
if (*p == '=') break; /* padding marks the end */
const char *hit = strchr(B32_ALPHABET, *p);
if (!hit) goto invalid;
buffer = (buffer << 5) | (unsigned)(hit - B32_ALPHABET);
bits += 5;
if (bits >= 8) {
bits -= 8;
buf_push(&out, (unsigned char)((buffer >> bits) & 0xff));
buffer &= (1u << bits) - 1u; /* keep only the leftover bits */
}
}
*out_len = out.len;
return out.data;
invalid:
free(out.data);
return NULL;
}
/* ---------------------------------------------------------------------------
* Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
* preserved).
* ------------------------------------------------------------------------ */
static void encode58(buf_t *out, const unsigned char *data, size_t len) {
/* Count leading zero bytes — each maps to a leading '1'. */
size_t zeros = 0;
while (zeros < len && data[zeros] == 0) zeros++;
/* Big-endian byte array (skipping the leading zeros) -> LE base-256. */
buf_t le = {0};
for (size_t i = zeros; i < len; i++) muladd_small(&le, 256, data[i]);
/* Base-convert to 58 digits (collected least-significant first; every
* digit is < 58, so they fit in single bytes). */
buf_t digits = {0};
while (le.len > 0) buf_push(&digits, (unsigned char)divmod_small(&le, 58));
for (size_t i = 0; i < zeros; i++) buf_push(out, '1');
for (size_t i = digits.len; i > 0; i--) buf_push(out, (unsigned char)B58_ALPHABET[digits.data[i - 1]]);
free(le.data);
free(digits.data);
}
static unsigned char *decode58(const char *s, size_t *out_len) {
const unsigned char *p = (const unsigned char *)s;
/* Count leading '1's — each maps to a 0x00 byte. */
size_t zeros = 0;
while (p[zeros] == '1') zeros++;
buf_t le = {0};
for (const unsigned char *q = p + zeros; *q; q++) {
const char *hit = strchr(B58_ALPHABET, *q);
if (!hit) {
free(le.data);
return NULL;
}
muladd_small(&le, 58, (unsigned)(hit - B58_ALPHABET));
}
/* LE -> minimal big-endian bytes (matches Go's big.Int.Bytes()). */
while (le.len > 0 && le.data[le.len - 1] == 0) le.len--; /* defensive strip */
buf_t out = {0};
buf_reserve(&out, 1);
for (size_t i = 0; i < zeros; i++) buf_push(&out, 0);
for (size_t i = le.len; i > 0; i--) buf_push(&out, le.data[i - 1]);
free(le.data);
*out_len = out.len;
return out.data;
}
/* ---------------------------------------------------------------------------
* Base62 — standard base-conversion of the byte array (no leading-zero
* special-casing beyond the standard big-int).
* ------------------------------------------------------------------------ */
static void encode62(buf_t *out, const unsigned char *data, size_t len) {
if (len == 0) return; /* empty input -> empty string */
buf_t le = {0};
for (size_t i = 0; i < len; i++) muladd_small(&le, 256, data[i]);
if (le.len == 0) { /* value zero */
buf_push(out, '0');
free(le.data);
return;
}
buf_t digits = {0};
while (le.len > 0) buf_push(&digits, (unsigned char)divmod_small(&le, 62));
for (size_t i = digits.len; i > 0; i--) buf_push(out, (unsigned char)B62_ALPHABET[digits.data[i - 1]]);
free(le.data);
free(digits.data);
}
static unsigned char *decode62(const char *s, size_t *out_len) {
buf_t le = {0};
for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
const char *hit = strchr(B62_ALPHABET, *p);
if (!hit) {
free(le.data);
return NULL;
}
muladd_small(&le, 62, (unsigned)(hit - B62_ALPHABET));
}
while (le.len > 0 && le.data[le.len - 1] == 0) le.len--; /* defensive strip */
buf_t out = {0};
buf_reserve(&out, 1);
for (size_t i = le.len; i > 0; i--) buf_push(&out, le.data[i - 1]);
free(le.data);
*out_len = out.len;
return out.data;
}
/* ---------------------------------------------------------------------------
* Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
* group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
* one fewer char than (bytes+1) would suggest; decode reverses, padding with
* 'u' (value 84).
* ------------------------------------------------------------------------ */
static void encode85(buf_t *out, const unsigned char *data, size_t len) {
size_t i = 0;
while (i < len) {
size_t n = len - i < 4 ? len - i : 4;
int is_full = n == 4;
unsigned b[4] = {0, 0, 0, 0};
for (size_t j = 0; j < n; j++) b[j] = data[i + j];
unsigned u = b[0] * 16777216u + b[1] * 65536u + b[2] * 256u + b[3];
i += 4;
if (is_full && u == 0) {
buf_push(out, 'z'); /* zero-group shorthand */
continue;
}
unsigned digits[5] = {0, 0, 0, 0, 0};
unsigned v = u;
for (int k = 4; k >= 0; k--) {
digits[k] = v % 85;
v /= 85;
}
size_t emit = is_full ? 5 : n + 1; /* n bytes -> n+1 chars */
for (size_t k = 0; k < emit; k++) buf_push(out, (unsigned char)(digits[k] + 33));
}
}
static unsigned char *decode85(const char *s, size_t *out_len) {
buf_t out = {0};
buf_reserve(&out, 1);
buf_t group = {0}; /* accumulated digit values (0..84) */
for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
unsigned c = *p;
if (c == 'z') {
/* 'z' is only valid at a group boundary (an empty accumulator). */
if (group.len != 0) goto invalid;
buf_push(&out, 0);
buf_push(&out, 0);
buf_push(&out, 0);
buf_push(&out, 0);
continue;
}
if (c < 33 || c > 117) goto invalid;
buf_push(&group, (unsigned char)(c - 33));
if (group.len == 5) {
unsigned long long v = 0;
for (size_t j = 0; j < 5; j++) v = v * 85 + group.data[j];
if (v > 0xFFFFFFFFull) goto invalid; /* a 5-char group must fit in 32 bits */
buf_push(&out, (unsigned char)((v >> 24) & 0xff));
buf_push(&out, (unsigned char)((v >> 16) & 0xff));
buf_push(&out, (unsigned char)((v >> 8) & 0xff));
buf_push(&out, (unsigned char)(v & 0xff));
group.len = 0;
}
}
/* Handle a partial final group (2-4 chars -> 1-3 bytes). */
if (group.len > 0) {
size_t m = group.len;
if (m < 2) goto invalid; /* a lone trailing char is malformed */
while (group.len < 5) buf_push(&group, 84); /* pad with 'u' */
unsigned long long v = 0;
for (size_t j = 0; j < 5; j++) v = v * 85 + group.data[j];
if (v > 0xFFFFFFFFull) goto invalid;
unsigned char all[4] = {
(unsigned char)((v >> 24) & 0xff),
(unsigned char)((v >> 16) & 0xff),
(unsigned char)((v >> 8) & 0xff),
(unsigned char)(v & 0xff),
};
for (size_t j = 0; j < m - 1; j++) buf_push(&out, all[j]);
}
free(group.data);
*out_len = out.len;
return out.data;
invalid:
free(out.data);
free(group.data);
return NULL;
}
/* ---------------------------------------------------------------------------
* Public API
* ------------------------------------------------------------------------ */
/* Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
* private `encodeBytes`. Returns a malloc'd NUL-terminated string. */
static char *encode_bytes(const unsigned char *data, size_t len, scheme_t scheme) {
buf_t out = {0};
switch (scheme) {
case SCHEME_BASE32: encode32(&out, data, len); break;
case SCHEME_BASE58: encode58(&out, data, len); break;
case SCHEME_BASE62: encode62(&out, data, len); break;
case SCHEME_BASE85: encode85(&out, data, len); break;
}
buf_push(&out, '\0');
return (char *)out.data;
}
/* Dispatch an encoded string to the chosen scheme's decoder. Returns the
* malloc'd raw bytes (length in *out_len), or NULL when the input is invalid
* or malformed (mirrors the TS `null`). Mirrors the Go twin's private
* `decodeBytes`. */
static unsigned char *decode_bytes(const char *encoded, scheme_t scheme, size_t *out_len) {
switch (scheme) {
case SCHEME_BASE32: return decode32(encoded, out_len);
case SCHEME_BASE58: return decode58(encoded, out_len);
case SCHEME_BASE62: return decode62(encoded, out_len);
case SCHEME_BASE85: return decode85(encoded, out_len);
}
return NULL;
}
/* Encode the UTF-8 bytes of `text` per `scheme`. Empty text -> "".
* Mirrors `Encode` in cli/base-encoder/base-encoder.go. */
char *base_encode(const char *text, scheme_t scheme) {
return encode_bytes((const unsigned char *)text, strlen(text), scheme);
}
/* Decode `encoded` back to raw bytes (the UTF-8 text is those bytes verbatim
* — Go's `string(data)`, which never fails). Invalid chars / malformed ->
* NULL (mirrors the Go twin's `errInvalid` and the TS lib's `null`).
* Mirrors `Decode` in cli/base-encoder/base-encoder.go. */
unsigned char *base_decode(const char *encoded, scheme_t scheme, size_t *out_len) {
return decode_bytes(encoded, scheme, out_len);
}
/* ---------------------------------------------------------------------------
* Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
* Run directly: `cc -std=c11 c.c && ./a.out`
* ------------------------------------------------------------------------ */
int main(void) {
/* C strings cannot carry an embedded NUL, so the zero-byte vectors use
* explicit-length byte arrays through encode_bytes (strlen would stop at
* the first 0x00 and encode an empty input instead). */
const unsigned char z1[1] = {0};
const unsigned char z3[3] = {0, 0, 'A'};
const unsigned char z4[4] = {0, 0, 0, 0};
const unsigned char z8[8] = {0, 0, 0, 0, 0, 0, 0, 0};
/* Base32 — known values + RFC 4648 padding + case sensitivity. */
char *e = base_encode("hello", SCHEME_BASE32);
assert(e && strcmp(e, "NBSWY3DP") == 0);
free(e);
e = base_encode("foo", SCHEME_BASE32);
assert(e && strcmp(e, "MZXW6===") == 0); /* 3 bytes -> 5 chars + 3 '=' */
free(e);
size_t n = 0;
unsigned char *d = base_decode("NBSWY3DP", SCHEME_BASE32, &n);
assert(d && n == 5 && memcmp(d, "hello", 5) == 0);
free(d);
assert(base_decode("nbswy3dp", SCHEME_BASE32, &n) == NULL); /* lowercase rejected */
/* Base58 — each leading 0x00 byte -> a leading '1'. */
e = encode_bytes(z1, 1, SCHEME_BASE58);
assert(e && strcmp(e, "1") == 0);
free(e);
e = encode_bytes(z3, 3, SCHEME_BASE58);
assert(e && strncmp(e, "11", 2) == 0);
free(e);
d = base_decode("1", SCHEME_BASE58, &n);
assert(d && n == 1 && d[0] == 0);
free(d);
e = encode_bytes(z3, 3, SCHEME_BASE58);
d = base_decode(e, SCHEME_BASE58, &n);
assert(d && n == 3 && memcmp(d, z3, 3) == 0);
free(e);
free(d);
/* Base62 — plain big-int base conversion (no leading-zero preservation). */
e = base_encode("A", SCHEME_BASE62);
assert(e && strcmp(e, "13") == 0); /* 1*62 + 3 */
free(e);
d = base_decode("13", SCHEME_BASE62, &n);
assert(d && n == 1 && d[0] == 'A');
free(d);
e = encode_bytes(z1, 1, SCHEME_BASE62);
assert(e && strcmp(e, "0") == 0);
free(e);
d = base_decode("0", SCHEME_BASE62, &n);
assert(d && n == 0); /* minimal rep of 0 is empty */
free(d);
/* Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection. */
e = base_encode("hello", SCHEME_BASE85);
assert(e && strcmp(e, "BOu!rDZ") == 0);
free(e);
e = encode_bytes(z4, 4, SCHEME_BASE85);
assert(e && strcmp(e, "z") == 0);
free(e);
e = encode_bytes(z8, 8, SCHEME_BASE85);
assert(e && strcmp(e, "zz") == 0);
free(e);
assert(base_decode("uuuuu", SCHEME_BASE85, &n) == NULL); /* group overflows 32 bits */
assert(base_decode("B", SCHEME_BASE85, &n) == NULL); /* lone trailing char */
/* Cross-scheme — empty, multibyte round-trip, and invalid rejection. */
const scheme_t schemes[4] = {SCHEME_BASE32, SCHEME_BASE58, SCHEME_BASE62, SCHEME_BASE85};
for (int s = 0; s < 4; s++) {
e = base_encode("", schemes[s]);
assert(e && strcmp(e, "") == 0);
free(e);
d = base_decode("", schemes[s], &n);
assert(d && n == 0);
free(d);
const char *mb = "CosmoDev \xF0\x9F\x9A\x80"; /* U+1F680 rocket, as UTF-8 bytes */
e = base_encode(mb, schemes[s]);
d = base_decode(e, schemes[s], &n);
assert(d && n == strlen(mb) && memcmp(d, mb, n) == 0);
free(e);
free(d);
assert(base_decode("~!not-valid!~", schemes[s], &n) == NULL); /* '~' outside every alphabet */
}
puts("ok");
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →