Skip to content

Base32 / Base58 / Base62 / Base85 Encoder — Rust source

Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
//! (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Base Encoder tool, ported
//!           from cli/base-encoder/base-encoder.go (the authoritative Go twin).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (encode always succeeds, decode
//!     returns `Result` — `Err` mirrors the Go twin's `errInvalid` / the TS
//!     lib's `null`).
//!   - Functionally equivalent to the Go twin: same inputs -> same outputs.
//!   - Self-contained: std only — no external crates, no `num-bigint`.
//!
//! Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
//! array, which overflows `u128` for inputs longer than a few bytes. The Go
//! twin leans on `math/big`; with no crate available we implement the same
//! idea with a little-endian base-256 byte vector and two primitives —
//! `divmod_small` (peel a base-N digit off the little end) and `muladd_small`
//! (reassemble a number from its base-N digits). These are the textbook
//! arbitrary-precision building blocks and keep the port dependency-free.

/// Selects a byte-array base encoding. Mirrors the Go twin's `Scheme` type
/// (and the TS `Scheme` union `'base32' | 'base58' | 'base62' | 'base85'`).
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Scheme {
    Base32,
    Base58,
    Base62,
    Base85,
}

/// Error returned when an encoded string contains a character outside the
/// scheme's alphabet or is otherwise malformed. Mirrors the Go twin's
/// `errInvalid` and the TS lib's `null` return from the internal decoders.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct InvalidInput;

const B32_ALPHABET: &str = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
const B58_ALPHABET: &str = "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
const B62_ALPHABET: &str = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";

/// Maps the size of a final (partial) 5-byte chunk to the number of data
/// characters it emits before '=' padding, per RFC 4648. Index = byte count
/// (0..=4). Matches `outLen = [0, 2, 4, 5, 7][chunk.length]` in the TS.
const OUT_LEN_32: [usize; 5] = [0, 2, 4, 5, 7];

// ---------------------------------------------------------------------------
// Arbitrary-precision primitives (base-256, little-endian). Used by Base58 and
// Base62 so the port stays dependency-free.
// ---------------------------------------------------------------------------

/// Divide a little-endian base-256 unsigned integer by a small `base`
/// (<= 256), storing the quotient back into `digits` (with high zero limbs
/// stripped) and returning the remainder. The long-division step used to peel
/// base-N digits off the little end during encoding.
fn divmod_small(digits: &mut Vec<u8>, base: u32) -> u32 {
    let mut rem: u32 = 0;
    for d in digits.iter_mut().rev() {
        let cur = rem * 256 + *d as u32;
        *d = (cur / base) as u8;
        rem = cur % base;
    }
    // Strip high (trailing in LE) zero limbs — keeps the representation minimal.
    while digits.last() == Some(&0) {
        digits.pop();
    }
    rem
}

/// Multiply a little-endian base-256 unsigned integer by `base` and add
/// `digit`, in place. The inverse of [`divmod_small`]: used to reassemble a
/// number from its base-N digits (processed most-significant first).
fn muladd_small(digits: &mut Vec<u8>, base: u32, digit: u32) {
    let mut carry = digit;
    for d in digits.iter_mut() {
        let cur = (*d as u32) * base + carry;
        *d = (cur & 0xff) as u8;
        carry = cur >> 8;
    }
    while carry > 0 {
        digits.push((carry & 0xff) as u8);
        carry >>= 8;
    }
}

/// Little-endian base-256 -> minimal big-endian bytes (the form the encoders
/// emit and the decoders reconstruct). `digits` already has no high zero limb,
/// so reversing yields a minimal representation that matches Go's
/// `big.Int.Bytes()`.
fn to_be_bytes(digits: &[u8]) -> Vec<u8> {
    let mut out: Vec<u8> = digits.iter().rev().copied().collect();
    // Defensive: strip any accidental leading zero (shouldn't happen, but the
    // Go twin guarantees minimal output so we match that contract exactly).
    while out.first() == Some(&0) {
        out.remove(0);
    }
    out
}

// ---------------------------------------------------------------------------
// Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
// ---------------------------------------------------------------------------

fn encode32(data: &[u8]) -> String {
    let mut out = String::new();
    let mut i = 0;
    while i < data.len() {
        let end = (i + 5).min(data.len());
        let chunk = &data[i..end];
        let mut b = [0u32; 5];
        for (j, &byte) in chunk.iter().enumerate() {
            b[j] = byte as u32;
        }
        // Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
        let digits = [
            (b[0] >> 3) & 0x1f,
            ((b[0] << 2) | (b[1] >> 6)) & 0x1f,
            (b[1] >> 1) & 0x1f,
            ((b[1] << 4) | (b[2] >> 4)) & 0x1f,
            ((b[2] << 1) | (b[3] >> 7)) & 0x1f,
            (b[3] >> 2) & 0x1f,
            ((b[3] << 3) | (b[4] >> 5)) & 0x1f,
            b[4] & 0x1f,
        ];
        let out_len = if chunk.len() == 5 { 8 } else { OUT_LEN_32[chunk.len()] };
        let alpha = B32_ALPHABET.as_bytes();
        for k in 0..out_len {
            out.push(alpha[digits[k] as usize] as char);
        }
        for _ in out_len..8 {
            out.push('=');
        }
        i += 5;
    }
    out
}

fn decode32(s: &str) -> Result<Vec<u8>, InvalidInput> {
    let mut out = Vec::new();
    let mut buffer: u32 = 0;
    let mut bits: u32 = 0;
    let alpha = B32_ALPHABET.as_bytes();
    for &c in s.as_bytes() {
        if c == b'=' {
            break; // padding marks the end
        }
        let idx = match alpha.iter().position(|&a| a == c) {
            Some(i) => i as u32,
            None => return Err(InvalidInput),
        };
        buffer = (buffer << 5) | idx;
        bits += 5;
        if bits >= 8 {
            bits -= 8;
            out.push(((buffer >> bits) & 0xff) as u8);
            buffer &= (1 << bits) - 1; // keep only the leftover bits
        }
    }
    Ok(out)
}

// ---------------------------------------------------------------------------
// Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count preserved).
// ---------------------------------------------------------------------------

fn encode58(data: &[u8]) -> String {
    // Count leading zero bytes — each maps to a leading '1'.
    let mut zeros = 0;
    while zeros < data.len() && data[zeros] == 0 {
        zeros += 1;
    }
    // Big-endian byte array (skipping the leading zeros) -> LE base-256.
    let mut le: Vec<u8> = Vec::new();
    for &b in &data[zeros..] {
        muladd_small(&mut le, 256, b as u32);
    }
    // Base-convert to 58 digits (collected least-significant first).
    let mut digits: Vec<u32> = Vec::new();
    while !le.is_empty() {
        digits.push(divmod_small(&mut le, 58));
    }
    let mut out = String::new();
    for _ in 0..zeros {
        out.push('1');
    }
    let alpha = B58_ALPHABET.as_bytes();
    for &d in digits.iter().rev() {
        out.push(alpha[d as usize] as char);
    }
    out
}

fn decode58(s: &str) -> Result<Vec<u8>, InvalidInput> {
    let bytes = s.as_bytes();
    // Count leading '1's — each maps to a 0x00 byte.
    let mut zeros = 0;
    while zeros < bytes.len() && bytes[zeros] == b'1' {
        zeros += 1;
    }
    let mut le: Vec<u8> = Vec::new();
    let alpha = B58_ALPHABET.as_bytes();
    for &c in &bytes[zeros..] {
        let idx = match alpha.iter().position(|&a| a == c) {
            Some(i) => i as u32,
            None => return Err(InvalidInput),
        };
        muladd_small(&mut le, 58, idx);
    }
    // LE -> minimal big-endian bytes.
    let mut out = vec![0u8; zeros];
    out.extend_from_slice(&to_be_bytes(&le));
    Ok(out)
}

// ---------------------------------------------------------------------------
// Base62 — standard base-conversion of the byte array (no leading-zero
// special-casing beyond the standard big-int).
// ---------------------------------------------------------------------------

fn encode62(data: &[u8]) -> String {
    if data.is_empty() {
        return String::new();
    }
    let mut le: Vec<u8> = Vec::new();
    for &b in data {
        muladd_small(&mut le, 256, b as u32);
    }
    if le.is_empty() {
        return "0".to_string(); // value zero
    }
    let mut digits: Vec<u32> = Vec::new();
    while !le.is_empty() {
        digits.push(divmod_small(&mut le, 62));
    }
    let mut out = String::new();
    let alpha = B62_ALPHABET.as_bytes();
    for &d in digits.iter().rev() {
        out.push(alpha[d as usize] as char);
    }
    out
}

fn decode62(s: &str) -> Result<Vec<u8>, InvalidInput> {
    if s.is_empty() {
        return Ok(Vec::new());
    }
    let mut le: Vec<u8> = Vec::new();
    let alpha = B62_ALPHABET.as_bytes();
    for &c in s.as_bytes() {
        let idx = match alpha.iter().position(|&a| a == c) {
            Some(i) => i as u32,
            None => return Err(InvalidInput),
        };
        muladd_small(&mut le, 62, idx);
    }
    Ok(to_be_bytes(&le))
}

// ---------------------------------------------------------------------------
// Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
// group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
// one fewer char than (bytes+1) would suggest; decode reverses, padding with
// 'u' (value 84).
// ---------------------------------------------------------------------------

fn encode85(data: &[u8]) -> String {
    let mut out = String::new();
    let mut i = 0;
    while i < data.len() {
        let end = (i + 4).min(data.len());
        let chunk = &data[i..end];
        let is_full = chunk.len() == 4;
        let mut b = [0u32; 4];
        for (j, &byte) in chunk.iter().enumerate() {
            b[j] = byte as u32;
        }
        let u = b[0] * 16777216 + b[1] * 65536 + b[2] * 256 + b[3];
        if is_full && u == 0 {
            out.push('z'); // zero-group shorthand
            i += 4;
            continue;
        }
        let mut digits = [0u32; 5];
        let mut v = u;
        for k in (0..5).rev() {
            digits[k] = v % 85;
            v /= 85;
        }
        let emit = if is_full { 5 } else { chunk.len() + 1 }; // n bytes -> n+1 chars
        for k in 0..emit {
            out.push((digits[k] + 33) as u8 as char);
        }
        i += 4;
    }
    out
}

fn decode85(s: &str) -> Result<Vec<u8>, InvalidInput> {
    let mut out = Vec::new();
    let mut group: Vec<u64> = Vec::with_capacity(5);
    for &c in s.as_bytes() {
        if c == b'z' {
            // 'z' is only valid at a group boundary (an empty accumulator).
            if !group.is_empty() {
                return Err(InvalidInput);
            }
            out.extend_from_slice(&[0, 0, 0, 0]);
            continue;
        }
        if c < 33 || c > 117 {
            return Err(InvalidInput);
        }
        group.push((c - 33) as u64);
        if group.len() == 5 {
            let mut v: u64 = 0;
            for &d in &group {
                v = v * 85 + d;
            }
            if v > 0xffffffff {
                return Err(InvalidInput); // a 5-char group must fit in 32 bits
            }
            out.extend_from_slice(&[
                ((v >> 24) & 0xff) as u8,
                ((v >> 16) & 0xff) as u8,
                ((v >> 8) & 0xff) as u8,
                (v & 0xff) as u8,
            ]);
            group.clear();
        }
    }
    // Handle a partial final group (2-4 chars -> 1-3 bytes).
    if !group.is_empty() {
        let m = group.len();
        if m < 2 {
            return Err(InvalidInput); // a lone trailing char is malformed
        }
        while group.len() < 5 {
            group.push(84); // pad with 'u'
        }
        let mut v: u64 = 0;
        for &d in &group {
            v = v * 85 + d;
        }
        if v > 0xffffffff {
            return Err(InvalidInput);
        }
        let all = [
            ((v >> 24) & 0xff) as u8,
            ((v >> 16) & 0xff) as u8,
            ((v >> 8) & 0xff) as u8,
            (v & 0xff) as u8,
        ];
        out.extend_from_slice(&all[..m - 1]);
    }
    Ok(out)
}

// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------

/// Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
/// private `encodeBytes`.
fn encode_bytes(data: &[u8], scheme: Scheme) -> String {
    match scheme {
        Scheme::Base32 => encode32(data),
        Scheme::Base58 => encode58(data),
        Scheme::Base62 => encode62(data),
        Scheme::Base85 => encode85(data),
    }
}

/// Dispatch an encoded string to the chosen scheme's decoder. An invalid or
/// malformed input yields `InvalidInput` (mirroring the TS `null`). Mirrors the
/// Go twin's private `decodeBytes`.
fn decode_bytes(encoded: &str, scheme: Scheme) -> Result<Vec<u8>, InvalidInput> {
    match scheme {
        Scheme::Base32 => decode32(encoded),
        Scheme::Base58 => decode58(encoded),
        Scheme::Base62 => decode62(encoded),
        Scheme::Base85 => decode85(encoded),
    }
}

/// Returns the chosen-scheme encoding of the UTF-8 bytes of `text`. Empty text
/// encodes to "". It is the Rust twin of `Encode` in
/// cli/base-encoder/base-encoder.go.
pub fn encode(text: &str, scheme: Scheme) -> String {
    encode_bytes(text.as_bytes(), scheme)
}

/// Reverses an encoded string back to UTF-8 text. Invalid characters or a
/// malformed structure yield `Err(InvalidInput)` — mirroring the Go twin's
/// `errInvalid` and the TS lib's `null`. It is the Rust twin of `Decode` in
/// cli/base-encoder/base-encoder.go.
///
/// The decoded bytes are interpreted as UTF-8; lossy decoding is used so a
/// structurally-valid-but-non-UTF-8 payload never produces a second error
/// (mirroring Go's `string(data)`, which never fails).
pub fn decode(encoded: &str, scheme: Scheme) -> Result<String, InvalidInput> {
    let bytes = decode_bytes(encoded, scheme)?;
    Ok(String::from_utf8_lossy(&bytes).into_owned())
}

// ---------- tests (showcase-only; the canonical suite lives in cli/) ----------
#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn base32_known_values_and_padding() {
        assert_eq!(encode("hello", Scheme::Base32), "NBSWY3DP");
        // 3 bytes -> 5 data chars + 3 '=' pads.
        assert_eq!(encode("foo", Scheme::Base32), "MZXW6===");
        assert_eq!(decode("NBSWY3DP", Scheme::Base32).unwrap(), "hello");
        // lowercase is not in the RFC 4648 alphabet
        assert!(decode("nbswy3dp", Scheme::Base32).is_err());
    }

    #[test]
    fn base58_preserves_leading_zero_bytes() {
        // each leading 0x00 byte -> a leading '1'
        assert_eq!(encode("\u{0}", Scheme::Base58), "1");
        assert!(encode("\u{0}\u{0}A", Scheme::Base58).starts_with("11"));
        assert_eq!(decode("1", Scheme::Base58).unwrap(), "\u{0}");
        // round-trip preserves the leading zero bytes exactly
        assert_eq!(
            decode(&encode("\u{0}\u{0}A", Scheme::Base58), Scheme::Base58).unwrap(),
            "\u{0}\u{0}A"
        );
    }

    #[test]
    fn base62_big_int_conversion() {
        assert_eq!(encode("A", Scheme::Base62), "13"); // 1*62 + 3
        assert_eq!(decode("13", Scheme::Base62).unwrap(), "A");
        assert_eq!(encode("\u{0}", Scheme::Base62), "0");
        // no leading-zero preservation: the minimal rep of 0 is empty
        assert_eq!(decode("0", Scheme::Base62).unwrap(), "");
    }

    #[test]
    fn base85_ascii85_shorthand_and_overflow() {
        assert_eq!(encode("hello", Scheme::Base85), "BOu!rDZ");
        assert_eq!(encode("\u{0}\u{0}\u{0}\u{0}", Scheme::Base85), "z"); // zero-group shorthand
        assert_eq!(
            encode("\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}", Scheme::Base85),
            "zz"
        );
        // a 5-char group must fit in 32 bits; "uuuuu" overflows
        assert!(decode("uuuuu", Scheme::Base85).is_err());
        // a lone trailing char is a malformed partial group
        assert!(decode("B", Scheme::Base85).is_err());
    }

    #[test]
    fn cross_scheme_round_trip_and_rejects_invalid() {
        let schemes = [Scheme::Base32, Scheme::Base58, Scheme::Base62, Scheme::Base85];
        for scheme in schemes {
            assert_eq!(encode("", scheme), "");
            assert_eq!(decode("", scheme).unwrap(), "");
            // multibyte UTF-8 round-trips through every scheme
            assert_eq!(
                decode(&encode("CosmoDev \u{1f680}", scheme), scheme).unwrap(),
                "CosmoDev \u{1f680}"
            );
            // '~' is outside every supported alphabet
            assert!(decode("~!not-valid!~", scheme).is_err());
        }
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →