Skip to content

Base32 / Base58 / Base62 / Base85 Encoder — Zig source

Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
//! (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
//! text.
//!
//! Language: Zig (0.13, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Base Encoder tool, ported
//!           from cli/base-encoder/base-encoder.go (the authoritative Go twin).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (encode always succeeds, decode
//!     returns error.InvalidInput for invalid or malformed input — mirroring
//!     the TS lib's `null` and the Go twin's `errInvalid`).
//!   - Functionally equivalent to the Go twin: same inputs -> same outputs.
//!   - Self-contained: std only.
//!
//! Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
//! array, which overflows any fixed-width integer for inputs longer than a
//! few bytes. std.math.big.int.Managed is the stdlib equivalent of Go's
//! math/big; like the Rust sibling we stay closer to the metal instead — a
//! little-endian base-256 ArrayList(u8) and two primitives: divmodSmall
//! (peel a base-N digit off the little end) and muladdSmall (reassemble a
//! number from its base-N digits).
//!
//! String note: Zig slices are untyped bytes, so decode returns the decoded
//! bytes verbatim — exactly Go's `string(data)` semantics (which never fails
//! and never mangles). Languages with validated string types (Rust/Python/
//! ...) decode lossily.

const std = @import("std");

/// Selects a byte-array base encoding. Mirrors the Go twin's `Scheme` type
/// (and the TS `Scheme` union "base32" | "base58" | "base62" | "base85").
pub const Scheme = enum { base32, base58, base62, base85 };

/// Raised when an encoded string contains a character outside the scheme's
/// alphabet or is otherwise malformed. Mirrors the Go twin's `errInvalid`
/// and the TS lib's `null` return from the internal decoders.
pub const Error = std.mem.Allocator.Error || error{InvalidInput};

const B32_ALPHABET = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
const B58_ALPHABET =
    "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
const B62_ALPHABET =
    "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";

/// Data characters emitted by a final (partial) 5-byte chunk before '='
/// padding, per RFC 4648. Index = byte count (0..4). Matches the TS `outLen`
/// table.
const OUT_LEN_32 = [5]usize{ 0, 2, 4, 5, 7 };

// ---------------------------------------------------------------------------
// Arbitrary-precision primitives (base-256, little-endian). Used by Base58
// and Base62 so the port stays dependency-free.
// ---------------------------------------------------------------------------

/// Divide a little-endian base-256 unsigned integer by a small `base`
/// (<= 256), storing the quotient back into `digits` (with high zero limbs
/// stripped) and returning the remainder. The long-division step used to
/// peel base-N digits off the little end during encoding.
fn divmodSmall(digits: *std.ArrayList(u8), base: u32) u32 {
    var rem: u32 = 0;
    var i = digits.items.len;
    while (i > 0) {
        i -= 1;
        const cur = rem * 256 + digits.items[i];
        digits.items[i] = @intCast(cur / base);
        rem = cur % base;
    }
    // Strip high (trailing in LE) zero limbs — keeps the representation
    // minimal.
    while (digits.items.len > 0 and digits.items[digits.items.len - 1] == 0) {
        digits.shrinkRetainingCapacity(digits.items.len - 1);
    }
    return rem;
}

/// Multiply a little-endian base-256 unsigned integer by `base` and add
/// `digit`, in place. The inverse of `divmodSmall`: reassembles a number
/// from its base-N digits (processed most-significant first).
fn muladdSmall(digits: *std.ArrayList(u8), base: u32, digit: u32) std.mem.Allocator.Error!void {
    var carry: u32 = digit;
    for (digits.items) |*d| {
        const cur = @as(u32, d.*) * base + carry;
        d.* = @intCast(cur & 0xff);
        carry = cur >> 8;
    }
    while (carry > 0) {
        try digits.append(@intCast(carry & 0xff));
        carry >>= 8;
    }
}

/// Little-endian base-256 -> minimal big-endian bytes (the form the encoders
/// emit and the decoders reconstruct). Strips any accidental leading zero so
/// the output matches Go's `big.Int.Bytes()` exactly.
fn toBeBytes(allocator: std.mem.Allocator, le: []const u8) std.mem.Allocator.Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var i = le.len;
    while (i > 0) {
        i -= 1;
        try out.append(le[i]);
    }
    while (out.items.len > 0 and out.items[0] == 0) {
        _ = out.orderedRemove(0);
    }
    return out.toOwnedSlice();
}

// ---------------------------------------------------------------------------
// Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
// ---------------------------------------------------------------------------

fn encode32(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var i: usize = 0;
    while (i < data.len) : (i += 5) {
        const end = @min(i + 5, data.len);
        const chunk = data[i..end];
        var b = [5]u32{ 0, 0, 0, 0, 0 };
        for (chunk, 0..) |byte, j| b[j] = byte;
        // Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
        const digits = [8]u32{
            (b[0] >> 3) & 0x1f,
            ((b[0] << 2) | (b[1] >> 6)) & 0x1f,
            (b[1] >> 1) & 0x1f,
            ((b[1] << 4) | (b[2] >> 4)) & 0x1f,
            ((b[2] << 1) | (b[3] >> 7)) & 0x1f,
            (b[3] >> 2) & 0x1f,
            ((b[3] << 3) | (b[4] >> 5)) & 0x1f,
            b[4] & 0x1f,
        };
        const out_len: usize = if (chunk.len == 5) 8 else OUT_LEN_32[chunk.len];
        var k: usize = 0;
        while (k < out_len) : (k += 1) try out.append(B32_ALPHABET[@intCast(digits[k])]);
        while (k < 8) : (k += 1) try out.append('=');
    }
    return out.toOwnedSlice();
}

fn decode32(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var buffer: u32 = 0;
    var bits: u32 = 0;
    for (s) |c| {
        if (c == '=') break; // padding marks the end
        const idx = std.mem.indexOfScalar(u8, B32_ALPHABET, c) orelse
            return error.InvalidInput;
        buffer = (buffer << 5) | @as(u32, @intCast(idx));
        bits += 5;
        if (bits >= 8) {
            bits -= 8;
            try out.append(@intCast((buffer >> @intCast(bits)) & 0xff));
            buffer &= (@as(u32, 1) << @intCast(bits)) - 1; // keep only the leftover bits
        }
    }
    return out.toOwnedSlice();
}

// ---------------------------------------------------------------------------
// Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
// preserved).
// ---------------------------------------------------------------------------

fn encode58(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
    // Count leading zero bytes — each maps to a leading '1'.
    var zeros: usize = 0;
    while (zeros < data.len and data[zeros] == 0) zeros += 1;
    // Big-endian byte array (skipping the leading zeros) -> LE base-256.
    var le = std.ArrayList(u8).init(allocator);
    defer le.deinit();
    for (data[zeros..]) |byte| try muladdSmall(&le, 256, byte);
    // Base-convert to 58 digits (collected least-significant first; every
    // digit is < 58, so they fit in single bytes).
    var digits = std.ArrayList(u8).init(allocator);
    defer digits.deinit();
    while (le.items.len > 0) {
        const rem = divmodSmall(&le, 58);
        try digits.append(@intCast(rem));
    }
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    try out.appendNTimes('1', zeros);
    var i = digits.items.len;
    while (i > 0) {
        i -= 1;
        try out.append(B58_ALPHABET[digits.items[i]]);
    }
    return out.toOwnedSlice();
}

fn decode58(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    // Count leading '1's — each maps to a 0x00 byte.
    var zeros: usize = 0;
    while (zeros < s.len and s[zeros] == '1') zeros += 1;
    var le = std.ArrayList(u8).init(allocator);
    defer le.deinit();
    for (s[zeros..]) |c| {
        const idx = std.mem.indexOfScalar(u8, B58_ALPHABET, c) orelse
            return error.InvalidInput;
        try muladdSmall(&le, 58, @intCast(idx));
    }
    // LE -> minimal big-endian bytes (matches Go's big.Int.Bytes()).
    const body = try toBeBytes(allocator, le.items);
    defer allocator.free(body);
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    try out.appendNTimes(0, zeros);
    try out.appendSlice(body);
    return out.toOwnedSlice();
}

// ---------------------------------------------------------------------------
// Base62 — standard base-conversion of the byte array (no leading-zero
// special-casing beyond the standard big-int).
// ---------------------------------------------------------------------------

fn encode62(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    if (data.len == 0) return out.toOwnedSlice(); // empty input -> empty string
    var le = std.ArrayList(u8).init(allocator);
    defer le.deinit();
    for (data) |byte| try muladdSmall(&le, 256, byte);
    if (le.items.len == 0) {
        try out.append('0'); // value zero
        return out.toOwnedSlice();
    }
    var digits = std.ArrayList(u8).init(allocator);
    defer digits.deinit();
    while (le.items.len > 0) {
        const rem = divmodSmall(&le, 62);
        try digits.append(@intCast(rem));
    }
    var i = digits.items.len;
    while (i > 0) {
        i -= 1;
        try out.append(B62_ALPHABET[digits.items[i]]);
    }
    return out.toOwnedSlice();
}

fn decode62(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    if (s.len == 0) return allocator.alloc(u8, 0);
    var le = std.ArrayList(u8).init(allocator);
    defer le.deinit();
    for (s) |c| {
        const idx = std.mem.indexOfScalar(u8, B62_ALPHABET, c) orelse
            return error.InvalidInput;
        try muladdSmall(&le, 62, @intCast(idx));
    }
    return toBeBytes(allocator, le.items);
}

// ---------------------------------------------------------------------------
// Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
// group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
// one fewer char than (bytes+1) would suggest; decode reverses, padding
// with 'u' (value 84).
// ---------------------------------------------------------------------------

fn encode85(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var i: usize = 0;
    while (i < data.len) : (i += 4) {
        const end = @min(i + 4, data.len);
        const chunk = data[i..end];
        const is_full = chunk.len == 4;
        var b = [4]u32{ 0, 0, 0, 0 };
        for (chunk, 0..) |byte, j| b[j] = byte;
        const u = b[0] * 16777216 + b[1] * 65536 + b[2] * 256 + b[3];
        if (is_full and u == 0) {
            try out.append('z'); // zero-group shorthand
            continue;
        }
        var digits = [5]u32{ 0, 0, 0, 0, 0 };
        var v = u;
        var k: usize = 5;
        while (k > 0) {
            k -= 1;
            digits[k] = v % 85;
            v /= 85;
        }
        const emit: usize = if (is_full) 5 else chunk.len + 1; // n bytes -> n+1 chars
        for (digits[0..emit]) |d| try out.append(@intCast(d + 33));
    }
    return out.toOwnedSlice();
}

fn decode85(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    var group = std.ArrayList(u8).init(allocator); // accumulated digit values (0..84)
    defer group.deinit();
    for (s) |c| {
        if (c == 'z') {
            // 'z' is only valid at a group boundary (an empty accumulator).
            if (group.items.len != 0) return error.InvalidInput;
            try out.appendSlice(&[4]u8{ 0, 0, 0, 0 });
            continue;
        }
        if (c < 33 or c > 117) return error.InvalidInput;
        try group.append(c - 33);
        if (group.items.len == 5) {
            var v: u64 = 0;
            for (group.items) |d| v = v * 85 + d;
            if (v > 0xffffffff) return error.InvalidInput; // must fit in 32 bits
            try out.appendSlice(&[4]u8{
                @intCast((v >> 24) & 0xff),
                @intCast((v >> 16) & 0xff),
                @intCast((v >> 8) & 0xff),
                @intCast(v & 0xff),
            });
            group.clearRetainingCapacity();
        }
    }
    // Handle a partial final group (2-4 chars -> 1-3 bytes).
    if (group.items.len > 0) {
        const m = group.items.len;
        if (m < 2) return error.InvalidInput; // a lone trailing char is malformed
        while (group.items.len < 5) try group.append(84); // pad with 'u'
        var v: u64 = 0;
        for (group.items) |d| v = v * 85 + d;
        if (v > 0xffffffff) return error.InvalidInput;
        const all = [4]u8{
            @intCast((v >> 24) & 0xff),
            @intCast((v >> 16) & 0xff),
            @intCast((v >> 8) & 0xff),
            @intCast(v & 0xff),
        };
        try out.appendSlice(all[0 .. m - 1]);
    }
    return out.toOwnedSlice();
}

// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------

/// Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
/// private `encodeBytes`.
fn encodeBytes(allocator: std.mem.Allocator, data: []const u8, scheme: Scheme) std.mem.Allocator.Error![]u8 {
    return switch (scheme) {
        .base32 => encode32(allocator, data),
        .base58 => encode58(allocator, data),
        .base62 => encode62(allocator, data),
        .base85 => encode85(allocator, data),
    };
}

/// Dispatch an encoded string to the chosen scheme's decoder. An invalid or
/// malformed input yields error.InvalidInput (mirroring the TS `null`).
/// Mirrors the Go twin's private `decodeBytes`.
fn decodeBytes(allocator: std.mem.Allocator, encoded: []const u8, scheme: Scheme) Error![]u8 {
    return switch (scheme) {
        .base32 => decode32(allocator, encoded),
        .base58 => decode58(allocator, encoded),
        .base62 => decode62(allocator, encoded),
        .base85 => decode85(allocator, encoded),
    };
}

/// Returns the chosen-scheme encoding of the UTF-8 bytes of `text` (a Zig
/// slice IS those bytes). Empty text encodes to "". It is the Zig twin of
/// `Encode` in cli/base-encoder/base-encoder.go. Caller owns the result.
pub fn encode(allocator: std.mem.Allocator, text: []const u8, scheme: Scheme) std.mem.Allocator.Error![]u8 {
    return encodeBytes(allocator, text, scheme);
}

/// Reverses an encoded string back to the decoded bytes (verbatim — Go's
/// `string(data)`, which never fails). Invalid characters or a malformed
/// structure yield error.InvalidInput — mirroring the Go twin's `errInvalid`
/// and the TS lib's `null`. It is the Zig twin of `Decode` in
/// cli/base-encoder/base-encoder.go. Caller owns the result.
pub fn decode(allocator: std.mem.Allocator, encoded: []const u8, scheme: Scheme) Error![]u8 {
    return decodeBytes(allocator, encoded, scheme);
}

// ---------------------------------------------------------------------------
// Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
// Run directly: `zig run zig.zig`
// ---------------------------------------------------------------------------
pub fn main() !void {
    var gpa = std.heap.GeneralPurposeAllocator(.{}){};
    defer _ = gpa.deinit();
    const allocator = gpa.allocator();
    const stdout = std.io.getStdOut().writer();

    const nul = "\u{0}";

    // Base32 — known values + RFC 4648 padding + case sensitivity.
    {
        const e = try encode(allocator, "hello", .base32);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("NBSWY3DP", e);
    }
    {
        // 3 bytes -> 5 data chars + 3 '=' pads.
        const e = try encode(allocator, "foo", .base32);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("MZXW6===", e);
    }
    {
        const d = try decode(allocator, "NBSWY3DP", .base32);
        defer allocator.free(d);
        try std.testing.expectEqualStrings("hello", d);
    }
    // lowercase is not in the RFC 4648 alphabet
    try std.testing.expectError(error.InvalidInput, decode(allocator, "nbswy3dp", .base32));

    // Base58 — each leading 0x00 byte -> a leading '1'.
    {
        const e = try encode(allocator, nul, .base58);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("1", e);
    }
    {
        const e = try encode(allocator, nul ++ nul ++ "A", .base58);
        defer allocator.free(e);
        try std.testing.expect(e.len >= 2 and e[0] == '1' and e[1] == '1');
    }
    {
        const d = try decode(allocator, "1", .base58);
        defer allocator.free(d);
        try std.testing.expectEqualStrings(nul, d);
    }
    {
        // round-trip preserves the leading zero bytes exactly
        const e = try encode(allocator, nul ++ nul ++ "A", .base58);
        defer allocator.free(e);
        const d = try decode(allocator, e, .base58);
        defer allocator.free(d);
        try std.testing.expectEqualStrings(nul ++ nul ++ "A", d);
    }

    // Base62 — plain big-int base conversion (no leading-zero preservation).
    {
        const e = try encode(allocator, "A", .base62); // 1*62 + 3
        defer allocator.free(e);
        try std.testing.expectEqualStrings("13", e);
    }
    {
        const d = try decode(allocator, "13", .base62);
        defer allocator.free(d);
        try std.testing.expectEqualStrings("A", d);
    }
    {
        const e = try encode(allocator, nul, .base62);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("0", e);
    }
    {
        // no leading-zero preservation: the minimal rep of 0 is empty
        const d = try decode(allocator, "0", .base62);
        defer allocator.free(d);
        try std.testing.expectEqualStrings("", d);
    }

    // Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection.
    {
        const e = try encode(allocator, "hello", .base85);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("BOu!rDZ", e);
    }
    {
        const e = try encode(allocator, nul ++ nul ++ nul ++ nul, .base85);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("z", e); // zero-group shorthand
    }
    {
        const e = try encode(allocator, nul ++ nul ++ nul ++ nul ++ nul ++ nul ++ nul ++ nul, .base85);
        defer allocator.free(e);
        try std.testing.expectEqualStrings("zz", e);
    }
    // a 5-char group must fit in 32 bits; "uuuuu" overflows
    try std.testing.expectError(error.InvalidInput, decode(allocator, "uuuuu", .base85));
    // a lone trailing char is a malformed partial group
    try std.testing.expectError(error.InvalidInput, decode(allocator, "B", .base85));

    // Cross-scheme — empty, multibyte round-trip, and invalid rejection.
    const schemes = [_]Scheme{ .base32, .base58, .base62, .base85 };
    for (schemes) |scheme| {
        {
            const e = try encode(allocator, "", scheme);
            defer allocator.free(e);
            try std.testing.expectEqualStrings("", e);
        }
        {
            const d = try decode(allocator, "", scheme);
            defer allocator.free(d);
            try std.testing.expectEqualStrings("", d);
        }
        {
            // multibyte UTF-8 round-trips through every scheme
            const e = try encode(allocator, "CosmoDev \u{1f680}", scheme);
            defer allocator.free(e);
            const d = try decode(allocator, e, scheme);
            defer allocator.free(d);
            try std.testing.expectEqualStrings("CosmoDev \u{1f680}", d);
        }
        // '~' is outside every supported alphabet
        try std.testing.expectError(error.InvalidInput, decode(allocator, "~!not-valid!~", scheme));
    }

    try stdout.print("ok\n", .{});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →