Base32 / Base58 / Base62 / Base85 Encoder — Rust source
Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
//! (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the Base Encoder tool, ported
//! from cli/base-encoder/base-encoder.go (the authoritative Go twin).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (encode always succeeds, decode
//! returns `Result` — `Err` mirrors the Go twin's `errInvalid` / the TS
//! lib's `null`).
//! - Functionally equivalent to the Go twin: same inputs -> same outputs.
//! - Self-contained: std only — no external crates, no `num-bigint`.
//!
//! Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
//! array, which overflows `u128` for inputs longer than a few bytes. The Go
//! twin leans on `math/big`; with no crate available we implement the same
//! idea with a little-endian base-256 byte vector and two primitives —
//! `divmod_small` (peel a base-N digit off the little end) and `muladd_small`
//! (reassemble a number from its base-N digits). These are the textbook
//! arbitrary-precision building blocks and keep the port dependency-free.
/// Selects a byte-array base encoding. Mirrors the Go twin's `Scheme` type
/// (and the TS `Scheme` union `'base32' | 'base58' | 'base62' | 'base85'`).
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Scheme {
Base32,
Base58,
Base62,
Base85,
}
/// Error returned when an encoded string contains a character outside the
/// scheme's alphabet or is otherwise malformed. Mirrors the Go twin's
/// `errInvalid` and the TS lib's `null` return from the internal decoders.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct InvalidInput;
const B32_ALPHABET: &str = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
const B58_ALPHABET: &str = "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz";
const B62_ALPHABET: &str = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
/// Maps the size of a final (partial) 5-byte chunk to the number of data
/// characters it emits before '=' padding, per RFC 4648. Index = byte count
/// (0..=4). Matches `outLen = [0, 2, 4, 5, 7][chunk.length]` in the TS.
const OUT_LEN_32: [usize; 5] = [0, 2, 4, 5, 7];
// ---------------------------------------------------------------------------
// Arbitrary-precision primitives (base-256, little-endian). Used by Base58 and
// Base62 so the port stays dependency-free.
// ---------------------------------------------------------------------------
/// Divide a little-endian base-256 unsigned integer by a small `base`
/// (<= 256), storing the quotient back into `digits` (with high zero limbs
/// stripped) and returning the remainder. The long-division step used to peel
/// base-N digits off the little end during encoding.
fn divmod_small(digits: &mut Vec<u8>, base: u32) -> u32 {
let mut rem: u32 = 0;
for d in digits.iter_mut().rev() {
let cur = rem * 256 + *d as u32;
*d = (cur / base) as u8;
rem = cur % base;
}
// Strip high (trailing in LE) zero limbs — keeps the representation minimal.
while digits.last() == Some(&0) {
digits.pop();
}
rem
}
/// Multiply a little-endian base-256 unsigned integer by `base` and add
/// `digit`, in place. The inverse of [`divmod_small`]: used to reassemble a
/// number from its base-N digits (processed most-significant first).
fn muladd_small(digits: &mut Vec<u8>, base: u32, digit: u32) {
let mut carry = digit;
for d in digits.iter_mut() {
let cur = (*d as u32) * base + carry;
*d = (cur & 0xff) as u8;
carry = cur >> 8;
}
while carry > 0 {
digits.push((carry & 0xff) as u8);
carry >>= 8;
}
}
/// Little-endian base-256 -> minimal big-endian bytes (the form the encoders
/// emit and the decoders reconstruct). `digits` already has no high zero limb,
/// so reversing yields a minimal representation that matches Go's
/// `big.Int.Bytes()`.
fn to_be_bytes(digits: &[u8]) -> Vec<u8> {
let mut out: Vec<u8> = digits.iter().rev().copied().collect();
// Defensive: strip any accidental leading zero (shouldn't happen, but the
// Go twin guarantees minimal output so we match that contract exactly).
while out.first() == Some(&0) {
out.remove(0);
}
out
}
// ---------------------------------------------------------------------------
// Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
// ---------------------------------------------------------------------------
fn encode32(data: &[u8]) -> String {
let mut out = String::new();
let mut i = 0;
while i < data.len() {
let end = (i + 5).min(data.len());
let chunk = &data[i..end];
let mut b = [0u32; 5];
for (j, &byte) in chunk.iter().enumerate() {
b[j] = byte as u32;
}
// Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
let digits = [
(b[0] >> 3) & 0x1f,
((b[0] << 2) | (b[1] >> 6)) & 0x1f,
(b[1] >> 1) & 0x1f,
((b[1] << 4) | (b[2] >> 4)) & 0x1f,
((b[2] << 1) | (b[3] >> 7)) & 0x1f,
(b[3] >> 2) & 0x1f,
((b[3] << 3) | (b[4] >> 5)) & 0x1f,
b[4] & 0x1f,
];
let out_len = if chunk.len() == 5 { 8 } else { OUT_LEN_32[chunk.len()] };
let alpha = B32_ALPHABET.as_bytes();
for k in 0..out_len {
out.push(alpha[digits[k] as usize] as char);
}
for _ in out_len..8 {
out.push('=');
}
i += 5;
}
out
}
fn decode32(s: &str) -> Result<Vec<u8>, InvalidInput> {
let mut out = Vec::new();
let mut buffer: u32 = 0;
let mut bits: u32 = 0;
let alpha = B32_ALPHABET.as_bytes();
for &c in s.as_bytes() {
if c == b'=' {
break; // padding marks the end
}
let idx = match alpha.iter().position(|&a| a == c) {
Some(i) => i as u32,
None => return Err(InvalidInput),
};
buffer = (buffer << 5) | idx;
bits += 5;
if bits >= 8 {
bits -= 8;
out.push(((buffer >> bits) & 0xff) as u8);
buffer &= (1 << bits) - 1; // keep only the leftover bits
}
}
Ok(out)
}
// ---------------------------------------------------------------------------
// Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count preserved).
// ---------------------------------------------------------------------------
fn encode58(data: &[u8]) -> String {
// Count leading zero bytes — each maps to a leading '1'.
let mut zeros = 0;
while zeros < data.len() && data[zeros] == 0 {
zeros += 1;
}
// Big-endian byte array (skipping the leading zeros) -> LE base-256.
let mut le: Vec<u8> = Vec::new();
for &b in &data[zeros..] {
muladd_small(&mut le, 256, b as u32);
}
// Base-convert to 58 digits (collected least-significant first).
let mut digits: Vec<u32> = Vec::new();
while !le.is_empty() {
digits.push(divmod_small(&mut le, 58));
}
let mut out = String::new();
for _ in 0..zeros {
out.push('1');
}
let alpha = B58_ALPHABET.as_bytes();
for &d in digits.iter().rev() {
out.push(alpha[d as usize] as char);
}
out
}
fn decode58(s: &str) -> Result<Vec<u8>, InvalidInput> {
let bytes = s.as_bytes();
// Count leading '1's — each maps to a 0x00 byte.
let mut zeros = 0;
while zeros < bytes.len() && bytes[zeros] == b'1' {
zeros += 1;
}
let mut le: Vec<u8> = Vec::new();
let alpha = B58_ALPHABET.as_bytes();
for &c in &bytes[zeros..] {
let idx = match alpha.iter().position(|&a| a == c) {
Some(i) => i as u32,
None => return Err(InvalidInput),
};
muladd_small(&mut le, 58, idx);
}
// LE -> minimal big-endian bytes.
let mut out = vec![0u8; zeros];
out.extend_from_slice(&to_be_bytes(&le));
Ok(out)
}
// ---------------------------------------------------------------------------
// Base62 — standard base-conversion of the byte array (no leading-zero
// special-casing beyond the standard big-int).
// ---------------------------------------------------------------------------
fn encode62(data: &[u8]) -> String {
if data.is_empty() {
return String::new();
}
let mut le: Vec<u8> = Vec::new();
for &b in data {
muladd_small(&mut le, 256, b as u32);
}
if le.is_empty() {
return "0".to_string(); // value zero
}
let mut digits: Vec<u32> = Vec::new();
while !le.is_empty() {
digits.push(divmod_small(&mut le, 62));
}
let mut out = String::new();
let alpha = B62_ALPHABET.as_bytes();
for &d in digits.iter().rev() {
out.push(alpha[d as usize] as char);
}
out
}
fn decode62(s: &str) -> Result<Vec<u8>, InvalidInput> {
if s.is_empty() {
return Ok(Vec::new());
}
let mut le: Vec<u8> = Vec::new();
let alpha = B62_ALPHABET.as_bytes();
for &c in s.as_bytes() {
let idx = match alpha.iter().position(|&a| a == c) {
Some(i) => i as u32,
None => return Err(InvalidInput),
};
muladd_small(&mut le, 62, idx);
}
Ok(to_be_bytes(&le))
}
// ---------------------------------------------------------------------------
// Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
// group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
// one fewer char than (bytes+1) would suggest; decode reverses, padding with
// 'u' (value 84).
// ---------------------------------------------------------------------------
fn encode85(data: &[u8]) -> String {
let mut out = String::new();
let mut i = 0;
while i < data.len() {
let end = (i + 4).min(data.len());
let chunk = &data[i..end];
let is_full = chunk.len() == 4;
let mut b = [0u32; 4];
for (j, &byte) in chunk.iter().enumerate() {
b[j] = byte as u32;
}
let u = b[0] * 16777216 + b[1] * 65536 + b[2] * 256 + b[3];
if is_full && u == 0 {
out.push('z'); // zero-group shorthand
i += 4;
continue;
}
let mut digits = [0u32; 5];
let mut v = u;
for k in (0..5).rev() {
digits[k] = v % 85;
v /= 85;
}
let emit = if is_full { 5 } else { chunk.len() + 1 }; // n bytes -> n+1 chars
for k in 0..emit {
out.push((digits[k] + 33) as u8 as char);
}
i += 4;
}
out
}
fn decode85(s: &str) -> Result<Vec<u8>, InvalidInput> {
let mut out = Vec::new();
let mut group: Vec<u64> = Vec::with_capacity(5);
for &c in s.as_bytes() {
if c == b'z' {
// 'z' is only valid at a group boundary (an empty accumulator).
if !group.is_empty() {
return Err(InvalidInput);
}
out.extend_from_slice(&[0, 0, 0, 0]);
continue;
}
if c < 33 || c > 117 {
return Err(InvalidInput);
}
group.push((c - 33) as u64);
if group.len() == 5 {
let mut v: u64 = 0;
for &d in &group {
v = v * 85 + d;
}
if v > 0xffffffff {
return Err(InvalidInput); // a 5-char group must fit in 32 bits
}
out.extend_from_slice(&[
((v >> 24) & 0xff) as u8,
((v >> 16) & 0xff) as u8,
((v >> 8) & 0xff) as u8,
(v & 0xff) as u8,
]);
group.clear();
}
}
// Handle a partial final group (2-4 chars -> 1-3 bytes).
if !group.is_empty() {
let m = group.len();
if m < 2 {
return Err(InvalidInput); // a lone trailing char is malformed
}
while group.len() < 5 {
group.push(84); // pad with 'u'
}
let mut v: u64 = 0;
for &d in &group {
v = v * 85 + d;
}
if v > 0xffffffff {
return Err(InvalidInput);
}
let all = [
((v >> 24) & 0xff) as u8,
((v >> 16) & 0xff) as u8,
((v >> 8) & 0xff) as u8,
(v & 0xff) as u8,
];
out.extend_from_slice(&all[..m - 1]);
}
Ok(out)
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/// Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go twin's
/// private `encodeBytes`.
fn encode_bytes(data: &[u8], scheme: Scheme) -> String {
match scheme {
Scheme::Base32 => encode32(data),
Scheme::Base58 => encode58(data),
Scheme::Base62 => encode62(data),
Scheme::Base85 => encode85(data),
}
}
/// Dispatch an encoded string to the chosen scheme's decoder. An invalid or
/// malformed input yields `InvalidInput` (mirroring the TS `null`). Mirrors the
/// Go twin's private `decodeBytes`.
fn decode_bytes(encoded: &str, scheme: Scheme) -> Result<Vec<u8>, InvalidInput> {
match scheme {
Scheme::Base32 => decode32(encoded),
Scheme::Base58 => decode58(encoded),
Scheme::Base62 => decode62(encoded),
Scheme::Base85 => decode85(encoded),
}
}
/// Returns the chosen-scheme encoding of the UTF-8 bytes of `text`. Empty text
/// encodes to "". It is the Rust twin of `Encode` in
/// cli/base-encoder/base-encoder.go.
pub fn encode(text: &str, scheme: Scheme) -> String {
encode_bytes(text.as_bytes(), scheme)
}
/// Reverses an encoded string back to UTF-8 text. Invalid characters or a
/// malformed structure yield `Err(InvalidInput)` — mirroring the Go twin's
/// `errInvalid` and the TS lib's `null`. It is the Rust twin of `Decode` in
/// cli/base-encoder/base-encoder.go.
///
/// The decoded bytes are interpreted as UTF-8; lossy decoding is used so a
/// structurally-valid-but-non-UTF-8 payload never produces a second error
/// (mirroring Go's `string(data)`, which never fails).
pub fn decode(encoded: &str, scheme: Scheme) -> Result<String, InvalidInput> {
let bytes = decode_bytes(encoded, scheme)?;
Ok(String::from_utf8_lossy(&bytes).into_owned())
}
// ---------- tests (showcase-only; the canonical suite lives in cli/) ----------
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn base32_known_values_and_padding() {
assert_eq!(encode("hello", Scheme::Base32), "NBSWY3DP");
// 3 bytes -> 5 data chars + 3 '=' pads.
assert_eq!(encode("foo", Scheme::Base32), "MZXW6===");
assert_eq!(decode("NBSWY3DP", Scheme::Base32).unwrap(), "hello");
// lowercase is not in the RFC 4648 alphabet
assert!(decode("nbswy3dp", Scheme::Base32).is_err());
}
#[test]
fn base58_preserves_leading_zero_bytes() {
// each leading 0x00 byte -> a leading '1'
assert_eq!(encode("\u{0}", Scheme::Base58), "1");
assert!(encode("\u{0}\u{0}A", Scheme::Base58).starts_with("11"));
assert_eq!(decode("1", Scheme::Base58).unwrap(), "\u{0}");
// round-trip preserves the leading zero bytes exactly
assert_eq!(
decode(&encode("\u{0}\u{0}A", Scheme::Base58), Scheme::Base58).unwrap(),
"\u{0}\u{0}A"
);
}
#[test]
fn base62_big_int_conversion() {
assert_eq!(encode("A", Scheme::Base62), "13"); // 1*62 + 3
assert_eq!(decode("13", Scheme::Base62).unwrap(), "A");
assert_eq!(encode("\u{0}", Scheme::Base62), "0");
// no leading-zero preservation: the minimal rep of 0 is empty
assert_eq!(decode("0", Scheme::Base62).unwrap(), "");
}
#[test]
fn base85_ascii85_shorthand_and_overflow() {
assert_eq!(encode("hello", Scheme::Base85), "BOu!rDZ");
assert_eq!(encode("\u{0}\u{0}\u{0}\u{0}", Scheme::Base85), "z"); // zero-group shorthand
assert_eq!(
encode("\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}\u{0}", Scheme::Base85),
"zz"
);
// a 5-char group must fit in 32 bits; "uuuuu" overflows
assert!(decode("uuuuu", Scheme::Base85).is_err());
// a lone trailing char is a malformed partial group
assert!(decode("B", Scheme::Base85).is_err());
}
#[test]
fn cross_scheme_round_trip_and_rejects_invalid() {
let schemes = [Scheme::Base32, Scheme::Base58, Scheme::Base62, Scheme::Base85];
for scheme in schemes {
assert_eq!(encode("", scheme), "");
assert_eq!(decode("", scheme).unwrap(), "");
// multibyte UTF-8 round-trips through every scheme
assert_eq!(
decode(&encode("CosmoDev \u{1f680}", scheme), scheme).unwrap(),
"CosmoDev \u{1f680}"
);
// '~' is outside every supported alphabet
assert!(decode("~!not-valid!~", scheme).is_err());
}
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →