Base32 / Base58 / Base62 / Base85 Encoder — Zig source
Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
//! (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
//! text.
//!
//! Language: Zig (0.13, standard library only)
//! Source: CosmoDev polyglot showcase port of the Base Encoder tool, ported
//! from cli/base-encoder/base-encoder.go (the authoritative Go twin).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (encode always succeeds, decode
//! returns error.InvalidInput for invalid or malformed input — mirroring
//! the TS lib's `null` and the Go twin's `errInvalid`).
//! - Functionally equivalent to the Go twin: same inputs -> same outputs.
//! - Self-contained: std only.
//!
//! Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
//! array, which overflows any fixed-width integer for inputs longer than a
//! few bytes. std.math.big.int.Managed is the stdlib equivalent of Go's
//! math/big; like the Rust sibling we stay closer to the metal instead — a
//! little-endian base-256 ArrayList(u8) and two primitives: divmodSmall
//! (peel a base-N digit off the little end) and muladdSmall (reassemble a
//! number from its base-N digits).
//!
//! String note: Zig slices are untyped bytes, so decode returns the decoded
//! bytes verbatim — exactly Go's `string(data)` semantics (which never fails
//! and never mangles). Languages with validated string types (Rust/Python/
//! ...) decode lossily.
const std = @import("std");
/// Selects a byte-array base encoding. Mirrors the Go twin's `Scheme` type
/// (and the TS `Scheme` union "base32" | "base58" | "base62" | "base85").
pub const Scheme = enum { base32, base58, base62, base85 };
/// Raised when an encoded string contains a character outside the scheme's
/// alphabet or is otherwise malformed. Mirrors the Go twin's `errInvalid`
/// and the TS lib's `null` return from the internal decoders.
pub const Error = std.mem.Allocator.Error || error{InvalidInput};
const B32_ALPHABET = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
const B58_ALPHABET =
"123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
const B62_ALPHABET =
"0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
/// Data characters emitted by a final (partial) 5-byte chunk before '='
/// padding, per RFC 4648. Index = byte count (0..4). Matches the TS `outLen`
/// table.
const OUT_LEN_32 = [5]usize{ 0, 2, 4, 5, 7 };
// ---------------------------------------------------------------------------
// Arbitrary-precision primitives (base-256, little-endian). Used by Base58
// and Base62 so the port stays dependency-free.
// ---------------------------------------------------------------------------
/// Divide a little-endian base-256 unsigned integer by a small `base`
/// (<= 256), storing the quotient back into `digits` (with high zero limbs
/// stripped) and returning the remainder. The long-division step used to
/// peel base-N digits off the little end during encoding.
fn divmodSmall(digits: *std.ArrayList(u8), base: u32) u32 {
var rem: u32 = 0;
var i = digits.items.len;
while (i > 0) {
i -= 1;
const cur = rem * 256 + digits.items[i];
digits.items[i] = @intCast(cur / base);
rem = cur % base;
}
// Strip high (trailing in LE) zero limbs — keeps the representation
// minimal.
while (digits.items.len > 0 and digits.items[digits.items.len - 1] == 0) {
digits.shrinkRetainingCapacity(digits.items.len - 1);
}
return rem;
}
/// Multiply a little-endian base-256 unsigned integer by `base` and add
/// `digit`, in place. The inverse of `divmodSmall`: reassembles a number
/// from its base-N digits (processed most-significant first).
fn muladdSmall(digits: *std.ArrayList(u8), base: u32, digit: u32) std.mem.Allocator.Error!void {
var carry: u32 = digit;
for (digits.items) |*d| {
const cur = @as(u32, d.*) * base + carry;
d.* = @intCast(cur & 0xff);
carry = cur >> 8;
}
while (carry > 0) {
try digits.append(@intCast(carry & 0xff));
carry >>= 8;
}
}
/// Little-endian base-256 -> minimal big-endian bytes (the form the encoders
/// emit and the decoders reconstruct). Strips any accidental leading zero so
/// the output matches Go's `big.Int.Bytes()` exactly.
fn toBeBytes(allocator: std.mem.Allocator, le: []const u8) std.mem.Allocator.Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var i = le.len;
while (i > 0) {
i -= 1;
try out.append(le[i]);
}
while (out.items.len > 0 and out.items[0] == 0) {
_ = out.orderedRemove(0);
}
return out.toOwnedSlice();
}
// ---------------------------------------------------------------------------
// Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
// ---------------------------------------------------------------------------
fn encode32(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var i: usize = 0;
while (i < data.len) : (i += 5) {
const end = @min(i + 5, data.len);
const chunk = data[i..end];
var b = [5]u32{ 0, 0, 0, 0, 0 };
for (chunk, 0..) |byte, j| b[j] = byte;
// Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
const digits = [8]u32{
(b[0] >> 3) & 0x1f,
((b[0] << 2) | (b[1] >> 6)) & 0x1f,
(b[1] >> 1) & 0x1f,
((b[1] << 4) | (b[2] >> 4)) & 0x1f,
((b[2] << 1) | (b[3] >> 7)) & 0x1f,
(b[3] >> 2) & 0x1f,
((b[3] << 3) | (b[4] >> 5)) & 0x1f,
b[4] & 0x1f,
};
const out_len: usize = if (chunk.len == 5) 8 else OUT_LEN_32[chunk.len];
var k: usize = 0;
while (k < out_len) : (k += 1) try out.append(B32_ALPHABET[@intCast(digits[k])]);
while (k < 8) : (k += 1) try out.append('=');
}
return out.toOwnedSlice();
}
fn decode32(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var buffer: u32 = 0;
var bits: u32 = 0;
for (s) |c| {
if (c == '=') break; // padding marks the end
const idx = std.mem.indexOfScalar(u8, B32_ALPHABET, c) orelse
return error.InvalidInput;
buffer = (buffer << 5) | @as(u32, @intCast(idx));
bits += 5;
if (bits >= 8) {
bits -= 8;
try out.append(@intCast((buffer >> @intCast(bits)) & 0xff));
buffer &= (@as(u32, 1) << @intCast(bits)) - 1; // keep only the leftover bits
}
}
return out.toOwnedSlice();
}
// ---------------------------------------------------------------------------
// Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
// preserved).
// ---------------------------------------------------------------------------
fn encode58(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
// Count leading zero bytes — each maps to a leading '1'.
var zeros: usize = 0;
while (zeros < data.len and data[zeros] == 0) zeros += 1;
// Big-endian byte array (skipping the leading zeros) -> LE base-256.
var le = std.ArrayList(u8).init(allocator);
defer le.deinit();
for (data[zeros..]) |byte| try muladdSmall(&le, 256, byte);
// Base-convert to 58 digits (collected least-significant first; every
// digit is < 58, so they fit in single bytes).
var digits = std.ArrayList(u8).init(allocator);
defer digits.deinit();
while (le.items.len > 0) {
const rem = divmodSmall(&le, 58);
try digits.append(@intCast(rem));
}
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
try out.appendNTimes('1', zeros);
var i = digits.items.len;
while (i > 0) {
i -= 1;
try out.append(B58_ALPHABET[digits.items[i]]);
}
return out.toOwnedSlice();
}
fn decode58(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
// Count leading '1's — each maps to a 0x00 byte.
var zeros: usize = 0;
while (zeros < s.len and s[zeros] == '1') zeros += 1;
var le = std.ArrayList(u8).init(allocator);
defer le.deinit();
for (s[zeros..]) |c| {
const idx = std.mem.indexOfScalar(u8, B58_ALPHABET, c) orelse
return error.InvalidInput;
try muladdSmall(&le, 58, @intCast(idx));
}
// LE -> minimal big-endian bytes (matches Go's big.Int.Bytes()).
const body = try toBeBytes(allocator, le.items);
defer allocator.free(body);
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
try out.appendNTimes(0, zeros);
try out.appendSlice(body);
return out.toOwnedSlice();
}
// ---------------------------------------------------------------------------
// Base62 — standard base-conversion of the byte array (no leading-zero
// special-casing beyond the standard big-int).
// ---------------------------------------------------------------------------
fn encode62(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
if (data.len == 0) return out.toOwnedSlice(); // empty input -> empty string
var le = std.ArrayList(u8).init(allocator);
defer le.deinit();
for (data) |byte| try muladdSmall(&le, 256, byte);
if (le.items.len == 0) {
try out.append('0'); // value zero
return out.toOwnedSlice();
}
var digits = std.ArrayList(u8).init(allocator);
defer digits.deinit();
while (le.items.len > 0) {
const rem = divmodSmall(&le, 62);
try digits.append(@intCast(rem));
}
var i = digits.items.len;
while (i > 0) {
i -= 1;
try out.append(B62_ALPHABET[digits.items[i]]);
}
return out.toOwnedSlice();
}
fn decode62(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
if (s.len == 0) return allocator.alloc(u8, 0);
var le = std.ArrayList(u8).init(allocator);
defer le.deinit();
for (s) |c| {
const idx = std.mem.indexOfScalar(u8, B62_ALPHABET, c) orelse
return error.InvalidInput;
try muladdSmall(&le, 62, @intCast(idx));
}
return toBeBytes(allocator, le.items);
}
// ---------------------------------------------------------------------------
// Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
// group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
// one fewer char than (bytes+1) would suggest; decode reverses, padding
// with 'u' (value 84).
// ---------------------------------------------------------------------------
fn encode85(allocator: std.mem.Allocator, data: []const u8) std.mem.Allocator.Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var i: usize = 0;
while (i < data.len) : (i += 4) {
const end = @min(i + 4, data.len);
const chunk = data[i..end];
const is_full = chunk.len == 4;
var b = [4]u32{ 0, 0, 0, 0 };
for (chunk, 0..) |byte, j| b[j] = byte;
const u = b[0] * 16777216 + b[1] * 65536 + b[2] * 256 + b[3];
if (is_full and u == 0) {
try out.append('z'); // zero-group shorthand
continue;
}
var digits = [5]u32{ 0, 0, 0, 0, 0 };
var v = u;
var k: usize = 5;
while (k > 0) {
k -= 1;
digits[k] = v % 85;
v /= 85;
}
const emit: usize = if (is_full) 5 else chunk.len + 1; // n bytes -> n+1 chars
for (digits[0..emit]) |d| try out.append(@intCast(d + 33));
}
return out.toOwnedSlice();
}
fn decode85(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var group = std.ArrayList(u8).init(allocator); // accumulated digit values (0..84)
defer group.deinit();
for (s) |c| {
if (c == 'z') {
// 'z' is only valid at a group boundary (an empty accumulator).
if (group.items.len != 0) return error.InvalidInput;
try out.appendSlice(&[4]u8{ 0, 0, 0, 0 });
continue;
}
if (c < 33 or c > 117) return error.InvalidInput;
try group.append(c - 33);
if (group.items.len == 5) {
var v: u64 = 0;
for (group.items) |d| v = v * 85 + d;
if (v > 0xffffffff) return error.InvalidInput; // must fit in 32 bits
try out.appendSlice(&[4]u8{
@intCast((v >> 24) & 0xff),
@intCast((v >> 16) & 0xff),
@intCast((v >> 8) & 0xff),
@intCast(v & 0xff),
});
group.clearRetainingCapacity();
}
}
// Handle a partial final group (2-4 chars -> 1-3 bytes).
if (group.items.len > 0) {
const m = group.items.len;
if (m < 2) return error.InvalidInput; // a lone trailing char is malformed
while (group.items.len < 5) try group.append(84); // pad with 'u'
var v: u64 = 0;
for (group.items) |d| v = v * 85 + d;
if (v > 0xffffffff) return error.InvalidInput;
const all = [4]u8{
@intCast((v >> 24) & 0xff),
@intCast((v >> 16) & 0xff),
@intCast((v >> 8) & 0xff),
@intCast(v & 0xff),
};
try out.appendSlice(all[0 .. m - 1]);
}
return out.toOwnedSlice();
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/// Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
/// private `encodeBytes`.
fn encodeBytes(allocator: std.mem.Allocator, data: []const u8, scheme: Scheme) std.mem.Allocator.Error![]u8 {
return switch (scheme) {
.base32 => encode32(allocator, data),
.base58 => encode58(allocator, data),
.base62 => encode62(allocator, data),
.base85 => encode85(allocator, data),
};
}
/// Dispatch an encoded string to the chosen scheme's decoder. An invalid or
/// malformed input yields error.InvalidInput (mirroring the TS `null`).
/// Mirrors the Go twin's private `decodeBytes`.
fn decodeBytes(allocator: std.mem.Allocator, encoded: []const u8, scheme: Scheme) Error![]u8 {
return switch (scheme) {
.base32 => decode32(allocator, encoded),
.base58 => decode58(allocator, encoded),
.base62 => decode62(allocator, encoded),
.base85 => decode85(allocator, encoded),
};
}
/// Returns the chosen-scheme encoding of the UTF-8 bytes of `text` (a Zig
/// slice IS those bytes). Empty text encodes to "". It is the Zig twin of
/// `Encode` in cli/base-encoder/base-encoder.go. Caller owns the result.
pub fn encode(allocator: std.mem.Allocator, text: []const u8, scheme: Scheme) std.mem.Allocator.Error![]u8 {
return encodeBytes(allocator, text, scheme);
}
/// Reverses an encoded string back to the decoded bytes (verbatim — Go's
/// `string(data)`, which never fails). Invalid characters or a malformed
/// structure yield error.InvalidInput — mirroring the Go twin's `errInvalid`
/// and the TS lib's `null`. It is the Zig twin of `Decode` in
/// cli/base-encoder/base-encoder.go. Caller owns the result.
pub fn decode(allocator: std.mem.Allocator, encoded: []const u8, scheme: Scheme) Error![]u8 {
return decodeBytes(allocator, encoded, scheme);
}
// ---------------------------------------------------------------------------
// Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
// Run directly: `zig run zig.zig`
// ---------------------------------------------------------------------------
pub fn main() !void {
var gpa = std.heap.GeneralPurposeAllocator(.{}){};
defer _ = gpa.deinit();
const allocator = gpa.allocator();
const stdout = std.io.getStdOut().writer();
const nul = "\u{0}";
// Base32 — known values + RFC 4648 padding + case sensitivity.
{
const e = try encode(allocator, "hello", .base32);
defer allocator.free(e);
try std.testing.expectEqualStrings("NBSWY3DP", e);
}
{
// 3 bytes -> 5 data chars + 3 '=' pads.
const e = try encode(allocator, "foo", .base32);
defer allocator.free(e);
try std.testing.expectEqualStrings("MZXW6===", e);
}
{
const d = try decode(allocator, "NBSWY3DP", .base32);
defer allocator.free(d);
try std.testing.expectEqualStrings("hello", d);
}
// lowercase is not in the RFC 4648 alphabet
try std.testing.expectError(error.InvalidInput, decode(allocator, "nbswy3dp", .base32));
// Base58 — each leading 0x00 byte -> a leading '1'.
{
const e = try encode(allocator, nul, .base58);
defer allocator.free(e);
try std.testing.expectEqualStrings("1", e);
}
{
const e = try encode(allocator, nul ++ nul ++ "A", .base58);
defer allocator.free(e);
try std.testing.expect(e.len >= 2 and e[0] == '1' and e[1] == '1');
}
{
const d = try decode(allocator, "1", .base58);
defer allocator.free(d);
try std.testing.expectEqualStrings(nul, d);
}
{
// round-trip preserves the leading zero bytes exactly
const e = try encode(allocator, nul ++ nul ++ "A", .base58);
defer allocator.free(e);
const d = try decode(allocator, e, .base58);
defer allocator.free(d);
try std.testing.expectEqualStrings(nul ++ nul ++ "A", d);
}
// Base62 — plain big-int base conversion (no leading-zero preservation).
{
const e = try encode(allocator, "A", .base62); // 1*62 + 3
defer allocator.free(e);
try std.testing.expectEqualStrings("13", e);
}
{
const d = try decode(allocator, "13", .base62);
defer allocator.free(d);
try std.testing.expectEqualStrings("A", d);
}
{
const e = try encode(allocator, nul, .base62);
defer allocator.free(e);
try std.testing.expectEqualStrings("0", e);
}
{
// no leading-zero preservation: the minimal rep of 0 is empty
const d = try decode(allocator, "0", .base62);
defer allocator.free(d);
try std.testing.expectEqualStrings("", d);
}
// Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection.
{
const e = try encode(allocator, "hello", .base85);
defer allocator.free(e);
try std.testing.expectEqualStrings("BOu!rDZ", e);
}
{
const e = try encode(allocator, nul ++ nul ++ nul ++ nul, .base85);
defer allocator.free(e);
try std.testing.expectEqualStrings("z", e); // zero-group shorthand
}
{
const e = try encode(allocator, nul ++ nul ++ nul ++ nul ++ nul ++ nul ++ nul ++ nul, .base85);
defer allocator.free(e);
try std.testing.expectEqualStrings("zz", e);
}
// a 5-char group must fit in 32 bits; "uuuuu" overflows
try std.testing.expectError(error.InvalidInput, decode(allocator, "uuuuu", .base85));
// a lone trailing char is a malformed partial group
try std.testing.expectError(error.InvalidInput, decode(allocator, "B", .base85));
// Cross-scheme — empty, multibyte round-trip, and invalid rejection.
const schemes = [_]Scheme{ .base32, .base58, .base62, .base85 };
for (schemes) |scheme| {
{
const e = try encode(allocator, "", scheme);
defer allocator.free(e);
try std.testing.expectEqualStrings("", e);
}
{
const d = try decode(allocator, "", scheme);
defer allocator.free(d);
try std.testing.expectEqualStrings("", d);
}
{
// multibyte UTF-8 round-trips through every scheme
const e = try encode(allocator, "CosmoDev \u{1f680}", scheme);
defer allocator.free(e);
const d = try decode(allocator, e, scheme);
defer allocator.free(d);
try std.testing.expectEqualStrings("CosmoDev \u{1f680}", d);
}
// '~' is outside every supported alphabet
try std.testing.expectError(error.InvalidInput, decode(allocator, "~!not-valid!~", scheme));
}
try stdout.print("ok\n", .{});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →