Skip to content

Base32 / Base58 / Base62 / Base85 Encoder — C source

Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
 * (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
 * text.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the Base Encoder tool, ported
 *           from cli/base-encoder/base-encoder.go (the authoritative Go twin).
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never crashes (encode always succeeds, decode
 *     returns NULL for invalid or malformed input, mirroring the TS lib's
 *     `null` and the Go twin's `errInvalid`).
 *   - Functionally equivalent to the Go twin: same inputs -> same outputs.
 *   - Self-contained: stdlib only — no bignum library (the GMP/tommath
 *     dependency the Go twin avoids via math/big is not pulled in either).
 *
 * Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
 * array, which overflows any fixed-width integer for inputs longer than a
 * few bytes. The Go twin leans on math/big; with no stdlib bignum we
 * implement the same idea as the Rust sibling — a little-endian base-256
 * byte buffer and two primitives: divmod_small (peel a base-N digit off the
 * little end) and muladd_small (reassemble a number from its base-N digits).
 *
 * String note: C strings are byte arrays, so decode returns the raw bytes —
 * exactly Go's `string(data)` semantics (which never fails and never
 * mangles). Languages with validated string types (Rust/Python/...) decode
 * lossily; here the caller receives the bytes verbatim.
 */

#include <assert.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

/* One of the four supported byte-array base encodings. Mirrors the Go twin's
 * `Scheme` type and the TS `Scheme` union. */
typedef enum { SCHEME_BASE32, SCHEME_BASE58, SCHEME_BASE62, SCHEME_BASE85 } scheme_t;

static const char B32_ALPHABET[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
static const char B58_ALPHABET[] =
    "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
static const char B62_ALPHABET[] =
    "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";

/* Data characters emitted by a final (partial) 5-byte chunk before '='
 * padding, per RFC 4648. Index = byte count (0..4). Matches the TS `outLen`
 * table. */
static const int OUT_LEN_32[5] = {0, 2, 4, 5, 7};

/* ---------------------------------------------------------------------------
 * Growable byte buffer — used both for output strings and for the
 * little-endian base-256 bignum limbs. Aborts on OOM (showcase simplicity).
 * ------------------------------------------------------------------------ */

typedef struct {
    unsigned char *data;
    size_t len;
    size_t cap;
} buf_t;

static void buf_reserve(buf_t *b, size_t extra) {
    if (b->len + extra <= b->cap) return;
    size_t cap = b->cap ? b->cap : 16;
    while (cap < b->len + extra) cap *= 2;
    unsigned char *grown = realloc(b->data, cap);
    if (!grown) abort();
    b->data = grown;
    b->cap = cap;
}

static void buf_push(buf_t *b, unsigned char c) {
    buf_reserve(b, 1);
    b->data[b->len++] = c;
}

/* ---------------------------------------------------------------------------
 * Arbitrary-precision primitives (base-256, little-endian). Used by Base58
 * and Base62 so the port stays dependency-free.
 * ------------------------------------------------------------------------ */

/* Divide a little-endian base-256 unsigned integer by a small `base`
 * (<= 256), storing the quotient back in `le` (with high zero limbs
 * stripped) and returning the remainder. The long-division step used to
 * peel base-N digits off the little end during encoding. */
static unsigned divmod_small(buf_t *le, unsigned base) {
    unsigned rem = 0;
    for (size_t i = le->len; i > 0; i--) {
        unsigned cur = rem * 256 + le->data[i - 1];
        le->data[i - 1] = (unsigned char)(cur / base);
        rem = cur % base;
    }
    while (le->len > 0 && le->data[le->len - 1] == 0) le->len--;
    return rem;
}

/* Multiply a little-endian base-256 unsigned integer by `base` and add
 * `digit`, in place. The inverse of divmod_small: reassembles a number from
 * its base-N digits (processed most-significant first). */
static void muladd_small(buf_t *le, unsigned base, unsigned digit) {
    unsigned carry = digit;
    for (size_t i = 0; i < le->len; i++) {
        unsigned cur = (unsigned)le->data[i] * base + carry;
        le->data[i] = (unsigned char)(cur & 0xff);
        carry = cur >> 8;
    }
    while (carry > 0) {
        buf_push(le, (unsigned char)(carry & 0xff));
        carry >>= 8;
    }
}

/* ---------------------------------------------------------------------------
 * Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
 * ------------------------------------------------------------------------ */

static void encode32(buf_t *out, const unsigned char *data, size_t len) {
    size_t i = 0;
    while (i < len) {
        size_t n = len - i < 5 ? len - i : 5;
        unsigned b[5] = {0, 0, 0, 0, 0};
        for (size_t j = 0; j < n; j++) b[j] = data[i + j];
        /* Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian). */
        unsigned digits[8] = {
            (b[0] >> 3) & 0x1f,
            ((b[0] << 2) | (b[1] >> 6)) & 0x1f,
            (b[1] >> 1) & 0x1f,
            ((b[1] << 4) | (b[2] >> 4)) & 0x1f,
            ((b[2] << 1) | (b[3] >> 7)) & 0x1f,
            (b[3] >> 2) & 0x1f,
            ((b[3] << 3) | (b[4] >> 5)) & 0x1f,
            b[4] & 0x1f,
        };
        size_t out_len = n == 5 ? 8 : (size_t)OUT_LEN_32[n];
        for (size_t k = 0; k < out_len; k++) buf_push(out, (unsigned char)B32_ALPHABET[digits[k]]);
        for (size_t k = out_len; k < 8; k++) buf_push(out, '=');
        i += 5;
    }
}

static unsigned char *decode32(const char *s, size_t *out_len) {
    buf_t out = {0};
    buf_reserve(&out, 1); /* keep data non-NULL for the empty case */
    unsigned buffer = 0, bits = 0;
    for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
        if (*p == '=') break; /* padding marks the end */
        const char *hit = strchr(B32_ALPHABET, *p);
        if (!hit) goto invalid;
        buffer = (buffer << 5) | (unsigned)(hit - B32_ALPHABET);
        bits += 5;
        if (bits >= 8) {
            bits -= 8;
            buf_push(&out, (unsigned char)((buffer >> bits) & 0xff));
            buffer &= (1u << bits) - 1u; /* keep only the leftover bits */
        }
    }
    *out_len = out.len;
    return out.data;
invalid:
    free(out.data);
    return NULL;
}

/* ---------------------------------------------------------------------------
 * Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
 * preserved).
 * ------------------------------------------------------------------------ */

static void encode58(buf_t *out, const unsigned char *data, size_t len) {
    /* Count leading zero bytes — each maps to a leading '1'. */
    size_t zeros = 0;
    while (zeros < len && data[zeros] == 0) zeros++;
    /* Big-endian byte array (skipping the leading zeros) -> LE base-256. */
    buf_t le = {0};
    for (size_t i = zeros; i < len; i++) muladd_small(&le, 256, data[i]);
    /* Base-convert to 58 digits (collected least-significant first; every
     * digit is < 58, so they fit in single bytes). */
    buf_t digits = {0};
    while (le.len > 0) buf_push(&digits, (unsigned char)divmod_small(&le, 58));
    for (size_t i = 0; i < zeros; i++) buf_push(out, '1');
    for (size_t i = digits.len; i > 0; i--) buf_push(out, (unsigned char)B58_ALPHABET[digits.data[i - 1]]);
    free(le.data);
    free(digits.data);
}

static unsigned char *decode58(const char *s, size_t *out_len) {
    const unsigned char *p = (const unsigned char *)s;
    /* Count leading '1's — each maps to a 0x00 byte. */
    size_t zeros = 0;
    while (p[zeros] == '1') zeros++;
    buf_t le = {0};
    for (const unsigned char *q = p + zeros; *q; q++) {
        const char *hit = strchr(B58_ALPHABET, *q);
        if (!hit) {
            free(le.data);
            return NULL;
        }
        muladd_small(&le, 58, (unsigned)(hit - B58_ALPHABET));
    }
    /* LE -> minimal big-endian bytes (matches Go's big.Int.Bytes()). */
    while (le.len > 0 && le.data[le.len - 1] == 0) le.len--; /* defensive strip */
    buf_t out = {0};
    buf_reserve(&out, 1);
    for (size_t i = 0; i < zeros; i++) buf_push(&out, 0);
    for (size_t i = le.len; i > 0; i--) buf_push(&out, le.data[i - 1]);
    free(le.data);
    *out_len = out.len;
    return out.data;
}

/* ---------------------------------------------------------------------------
 * Base62 — standard base-conversion of the byte array (no leading-zero
 * special-casing beyond the standard big-int).
 * ------------------------------------------------------------------------ */

static void encode62(buf_t *out, const unsigned char *data, size_t len) {
    if (len == 0) return; /* empty input -> empty string */
    buf_t le = {0};
    for (size_t i = 0; i < len; i++) muladd_small(&le, 256, data[i]);
    if (le.len == 0) { /* value zero */
        buf_push(out, '0');
        free(le.data);
        return;
    }
    buf_t digits = {0};
    while (le.len > 0) buf_push(&digits, (unsigned char)divmod_small(&le, 62));
    for (size_t i = digits.len; i > 0; i--) buf_push(out, (unsigned char)B62_ALPHABET[digits.data[i - 1]]);
    free(le.data);
    free(digits.data);
}

static unsigned char *decode62(const char *s, size_t *out_len) {
    buf_t le = {0};
    for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
        const char *hit = strchr(B62_ALPHABET, *p);
        if (!hit) {
            free(le.data);
            return NULL;
        }
        muladd_small(&le, 62, (unsigned)(hit - B62_ALPHABET));
    }
    while (le.len > 0 && le.data[le.len - 1] == 0) le.len--; /* defensive strip */
    buf_t out = {0};
    buf_reserve(&out, 1);
    for (size_t i = le.len; i > 0; i--) buf_push(&out, le.data[i - 1]);
    free(le.data);
    *out_len = out.len;
    return out.data;
}

/* ---------------------------------------------------------------------------
 * Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
 * group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
 * one fewer char than (bytes+1) would suggest; decode reverses, padding with
 * 'u' (value 84).
 * ------------------------------------------------------------------------ */

static void encode85(buf_t *out, const unsigned char *data, size_t len) {
    size_t i = 0;
    while (i < len) {
        size_t n = len - i < 4 ? len - i : 4;
        int is_full = n == 4;
        unsigned b[4] = {0, 0, 0, 0};
        for (size_t j = 0; j < n; j++) b[j] = data[i + j];
        unsigned u = b[0] * 16777216u + b[1] * 65536u + b[2] * 256u + b[3];
        i += 4;
        if (is_full && u == 0) {
            buf_push(out, 'z'); /* zero-group shorthand */
            continue;
        }
        unsigned digits[5] = {0, 0, 0, 0, 0};
        unsigned v = u;
        for (int k = 4; k >= 0; k--) {
            digits[k] = v % 85;
            v /= 85;
        }
        size_t emit = is_full ? 5 : n + 1; /* n bytes -> n+1 chars */
        for (size_t k = 0; k < emit; k++) buf_push(out, (unsigned char)(digits[k] + 33));
    }
}

static unsigned char *decode85(const char *s, size_t *out_len) {
    buf_t out = {0};
    buf_reserve(&out, 1);
    buf_t group = {0}; /* accumulated digit values (0..84) */
    for (const unsigned char *p = (const unsigned char *)s; *p; p++) {
        unsigned c = *p;
        if (c == 'z') {
            /* 'z' is only valid at a group boundary (an empty accumulator). */
            if (group.len != 0) goto invalid;
            buf_push(&out, 0);
            buf_push(&out, 0);
            buf_push(&out, 0);
            buf_push(&out, 0);
            continue;
        }
        if (c < 33 || c > 117) goto invalid;
        buf_push(&group, (unsigned char)(c - 33));
        if (group.len == 5) {
            unsigned long long v = 0;
            for (size_t j = 0; j < 5; j++) v = v * 85 + group.data[j];
            if (v > 0xFFFFFFFFull) goto invalid; /* a 5-char group must fit in 32 bits */
            buf_push(&out, (unsigned char)((v >> 24) & 0xff));
            buf_push(&out, (unsigned char)((v >> 16) & 0xff));
            buf_push(&out, (unsigned char)((v >> 8) & 0xff));
            buf_push(&out, (unsigned char)(v & 0xff));
            group.len = 0;
        }
    }
    /* Handle a partial final group (2-4 chars -> 1-3 bytes). */
    if (group.len > 0) {
        size_t m = group.len;
        if (m < 2) goto invalid; /* a lone trailing char is malformed */
        while (group.len < 5) buf_push(&group, 84); /* pad with 'u' */
        unsigned long long v = 0;
        for (size_t j = 0; j < 5; j++) v = v * 85 + group.data[j];
        if (v > 0xFFFFFFFFull) goto invalid;
        unsigned char all[4] = {
            (unsigned char)((v >> 24) & 0xff),
            (unsigned char)((v >> 16) & 0xff),
            (unsigned char)((v >> 8) & 0xff),
            (unsigned char)(v & 0xff),
        };
        for (size_t j = 0; j < m - 1; j++) buf_push(&out, all[j]);
    }
    free(group.data);
    *out_len = out.len;
    return out.data;
invalid:
    free(out.data);
    free(group.data);
    return NULL;
}

/* ---------------------------------------------------------------------------
 * Public API
 * ------------------------------------------------------------------------ */

/* Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
 * private `encodeBytes`. Returns a malloc'd NUL-terminated string. */
static char *encode_bytes(const unsigned char *data, size_t len, scheme_t scheme) {
    buf_t out = {0};
    switch (scheme) {
    case SCHEME_BASE32: encode32(&out, data, len); break;
    case SCHEME_BASE58: encode58(&out, data, len); break;
    case SCHEME_BASE62: encode62(&out, data, len); break;
    case SCHEME_BASE85: encode85(&out, data, len); break;
    }
    buf_push(&out, '\0');
    return (char *)out.data;
}

/* Dispatch an encoded string to the chosen scheme's decoder. Returns the
 * malloc'd raw bytes (length in *out_len), or NULL when the input is invalid
 * or malformed (mirrors the TS `null`). Mirrors the Go twin's private
 * `decodeBytes`. */
static unsigned char *decode_bytes(const char *encoded, scheme_t scheme, size_t *out_len) {
    switch (scheme) {
    case SCHEME_BASE32: return decode32(encoded, out_len);
    case SCHEME_BASE58: return decode58(encoded, out_len);
    case SCHEME_BASE62: return decode62(encoded, out_len);
    case SCHEME_BASE85: return decode85(encoded, out_len);
    }
    return NULL;
}

/* Encode the UTF-8 bytes of `text` per `scheme`. Empty text -> "".
 * Mirrors `Encode` in cli/base-encoder/base-encoder.go. */
char *base_encode(const char *text, scheme_t scheme) {
    return encode_bytes((const unsigned char *)text, strlen(text), scheme);
}

/* Decode `encoded` back to raw bytes (the UTF-8 text is those bytes verbatim
 * — Go's `string(data)`, which never fails). Invalid chars / malformed ->
 * NULL (mirrors the Go twin's `errInvalid` and the TS lib's `null`).
 * Mirrors `Decode` in cli/base-encoder/base-encoder.go. */
unsigned char *base_decode(const char *encoded, scheme_t scheme, size_t *out_len) {
    return decode_bytes(encoded, scheme, out_len);
}

/* ---------------------------------------------------------------------------
 * Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
 * Run directly: `cc -std=c11 c.c && ./a.out`
 * ------------------------------------------------------------------------ */
int main(void) {
    /* C strings cannot carry an embedded NUL, so the zero-byte vectors use
     * explicit-length byte arrays through encode_bytes (strlen would stop at
     * the first 0x00 and encode an empty input instead). */
    const unsigned char z1[1] = {0};
    const unsigned char z3[3] = {0, 0, 'A'};
    const unsigned char z4[4] = {0, 0, 0, 0};
    const unsigned char z8[8] = {0, 0, 0, 0, 0, 0, 0, 0};

    /* Base32 — known values + RFC 4648 padding + case sensitivity. */
    char *e = base_encode("hello", SCHEME_BASE32);
    assert(e && strcmp(e, "NBSWY3DP") == 0);
    free(e);
    e = base_encode("foo", SCHEME_BASE32);
    assert(e && strcmp(e, "MZXW6===") == 0); /* 3 bytes -> 5 chars + 3 '=' */
    free(e);
    size_t n = 0;
    unsigned char *d = base_decode("NBSWY3DP", SCHEME_BASE32, &n);
    assert(d && n == 5 && memcmp(d, "hello", 5) == 0);
    free(d);
    assert(base_decode("nbswy3dp", SCHEME_BASE32, &n) == NULL); /* lowercase rejected */

    /* Base58 — each leading 0x00 byte -> a leading '1'. */
    e = encode_bytes(z1, 1, SCHEME_BASE58);
    assert(e && strcmp(e, "1") == 0);
    free(e);
    e = encode_bytes(z3, 3, SCHEME_BASE58);
    assert(e && strncmp(e, "11", 2) == 0);
    free(e);
    d = base_decode("1", SCHEME_BASE58, &n);
    assert(d && n == 1 && d[0] == 0);
    free(d);
    e = encode_bytes(z3, 3, SCHEME_BASE58);
    d = base_decode(e, SCHEME_BASE58, &n);
    assert(d && n == 3 && memcmp(d, z3, 3) == 0);
    free(e);
    free(d);

    /* Base62 — plain big-int base conversion (no leading-zero preservation). */
    e = base_encode("A", SCHEME_BASE62);
    assert(e && strcmp(e, "13") == 0); /* 1*62 + 3 */
    free(e);
    d = base_decode("13", SCHEME_BASE62, &n);
    assert(d && n == 1 && d[0] == 'A');
    free(d);
    e = encode_bytes(z1, 1, SCHEME_BASE62);
    assert(e && strcmp(e, "0") == 0);
    free(e);
    d = base_decode("0", SCHEME_BASE62, &n);
    assert(d && n == 0); /* minimal rep of 0 is empty */
    free(d);

    /* Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection. */
    e = base_encode("hello", SCHEME_BASE85);
    assert(e && strcmp(e, "BOu!rDZ") == 0);
    free(e);
    e = encode_bytes(z4, 4, SCHEME_BASE85);
    assert(e && strcmp(e, "z") == 0);
    free(e);
    e = encode_bytes(z8, 8, SCHEME_BASE85);
    assert(e && strcmp(e, "zz") == 0);
    free(e);
    assert(base_decode("uuuuu", SCHEME_BASE85, &n) == NULL); /* group overflows 32 bits */
    assert(base_decode("B", SCHEME_BASE85, &n) == NULL);     /* lone trailing char */

    /* Cross-scheme — empty, multibyte round-trip, and invalid rejection. */
    const scheme_t schemes[4] = {SCHEME_BASE32, SCHEME_BASE58, SCHEME_BASE62, SCHEME_BASE85};
    for (int s = 0; s < 4; s++) {
        e = base_encode("", schemes[s]);
        assert(e && strcmp(e, "") == 0);
        free(e);
        d = base_decode("", schemes[s], &n);
        assert(d && n == 0);
        free(d);
        const char *mb = "CosmoDev \xF0\x9F\x9A\x80"; /* U+1F680 rocket, as UTF-8 bytes */
        e = base_encode(mb, schemes[s]);
        d = base_decode(e, schemes[s], &n);
        assert(d && n == strlen(mb) && memcmp(d, mb, n) == 0);
        free(e);
        free(d);
        assert(base_decode("~!not-valid!~", schemes[s], &n) == NULL); /* '~' outside every alphabet */
    }

    puts("ok");
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →