Skip to content

Base32 / Base58 / Base62 / Base85 Encoder — Python source

Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
(Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input text.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Base Encoder tool, ported
          from cli/base-encoder/base-encoder.go (the authoritative Go twin).
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises (``decode`` returns ``None`` for invalid
    or malformed input, mirroring the TS lib's ``null`` and the Go twin's
    ``errInvalid``).
  - Functionally equivalent to the Go twin: same inputs -> same outputs.
  - Self-contained: stdlib only (no pip packages).

Arbitrary-precision note: Base58 and Base62 base-convert the whole byte array.
Python's built-in ``int`` is arbitrary-precision, so we get the exact same
semantics as the Go twin's ``math/big`` for free — no manual bignum code is
needed (unlike the dependency-free Rust/PHP ports).
"""

from __future__ import annotations

from typing import List, Literal, Optional

__all__ = ["encode", "decode"]

# One of the four supported byte-array base encodings. Mirrors the Go twin's
# `Scheme` type and the TS `Scheme` union.
Scheme = Literal["base32", "base58", "base62", "base85"]

B32_ALPHABET = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567"
B58_ALPHABET = "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz"
B62_ALPHABET = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"

# Data characters emitted by a final (partial) 5-byte chunk before '=' padding,
# per RFC 4648. Index = byte count (0..4). Matches the TS `outLen` table.
_OUT_LEN_32 = (0, 2, 4, 5, 7)


# ---------------------------------------------------------------------------
# Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
# ---------------------------------------------------------------------------


def _encode32(data: bytes) -> str:
    out: List[str] = []
    for i in range(0, len(data), 5):
        chunk = data[i : i + 5]
        b = [0, 0, 0, 0, 0]
        for j in range(len(chunk)):
            b[j] = chunk[j]
        # Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
        digits = [
            (b[0] >> 3) & 0x1F,
            ((b[0] << 2) | (b[1] >> 6)) & 0x1F,
            (b[1] >> 1) & 0x1F,
            ((b[1] << 4) | (b[2] >> 4)) & 0x1F,
            ((b[2] << 1) | (b[3] >> 7)) & 0x1F,
            (b[3] >> 2) & 0x1F,
            ((b[3] << 3) | (b[4] >> 5)) & 0x1F,
            b[4] & 0x1F,
        ]
        group = "".join(B32_ALPHABET[d] for d in digits)
        if len(chunk) < 5:
            # Truncate to the data chars and pad with '=' to 8 total.
            n = _OUT_LEN_32[len(chunk)]
            group = group[:n] + "=" * (8 - n)
        out.append(group)
    return "".join(out)


def _decode32(s: str) -> Optional[bytes]:
    out = bytearray()
    buffer = 0
    bits = 0
    for c in s:
        if c == "=":
            break  # padding marks the end
        idx = B32_ALPHABET.find(c)
        if idx == -1:
            return None
        buffer = (buffer << 5) | idx
        bits += 5
        if bits >= 8:
            bits -= 8
            out.append((buffer >> bits) & 0xFF)
            buffer &= (1 << bits) - 1  # keep only the leftover bits
    return bytes(out)


# ---------------------------------------------------------------------------
# Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count preserved).
# ---------------------------------------------------------------------------


def _encode58(data: bytes) -> str:
    # Count leading zero bytes — each maps to a leading '1'.
    zeros = 0
    while zeros < len(data) and data[zeros] == 0:
        zeros += 1
    # Big-endian byte array (skipping the leading zeros) -> int.
    num = 0
    for b in data[zeros:]:
        num = (num << 8) | b
    # Base-convert to 58 digits (collected least-significant first).
    digits: List[int] = []
    while num > 0:
        num, rem = divmod(num, 58)
        digits.append(rem)
    return "1" * zeros + "".join(B58_ALPHABET[d] for d in reversed(digits))


def _decode58(s: str) -> Optional[bytes]:
    # Count leading '1's — each maps to a 0x00 byte.
    zeros = 0
    while zeros < len(s) and s[zeros] == "1":
        zeros += 1
    num = 0
    for c in s[zeros:]:
        idx = B58_ALPHABET.find(c)
        if idx == -1:
            return None
        num = num * 58 + idx
    # int -> minimal big-endian bytes (matches Go's big.Int.Bytes()).
    body = num.to_bytes((num.bit_length() + 7) // 8, "big") if num > 0 else b""
    return b"\x00" * zeros + body


# ---------------------------------------------------------------------------
# Base62 — standard base-conversion of the byte array (no leading-zero
# special-casing beyond the standard big-int).
# ---------------------------------------------------------------------------


def _encode62(data: bytes) -> str:
    if not data:
        return ""
    num = 0
    for b in data:
        num = (num << 8) | b
    if num == 0:
        return "0"
    digits: List[int] = []
    while num > 0:
        num, rem = divmod(num, 62)
        digits.append(rem)
    return "".join(B62_ALPHABET[d] for d in reversed(digits))


def _decode62(s: str) -> Optional[bytes]:
    if not s:
        return b""
    num = 0
    for c in s:
        idx = B62_ALPHABET.find(c)
        if idx == -1:
            return None
        num = num * 62 + idx
    return num.to_bytes((num.bit_length() + 7) // 8, "big") if num > 0 else b""


# ---------------------------------------------------------------------------
# Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full 4-zero
# group is shortened to 'z'. No <~ ~> delimiters. Partial final groups emit
# one fewer char than (bytes+1) would suggest; decode reverses, padding with
# 'u' (value 84).
# ---------------------------------------------------------------------------


def _encode85(data: bytes) -> str:
    out: List[str] = []
    for i in range(0, len(data), 4):
        chunk = data[i : i + 4]
        is_full = len(chunk) == 4
        b = [0, 0, 0, 0]
        for j in range(len(chunk)):
            b[j] = chunk[j]
        u = b[0] * 16777216 + b[1] * 65536 + b[2] * 256 + b[3]
        if is_full and u == 0:
            out.append("z")  # zero-group shorthand
            continue
        digits = [0, 0, 0, 0, 0]
        v = u
        for k in range(4, -1, -1):
            digits[k] = v % 85
            v //= 85
        chars = "".join(chr(d + 33) for d in digits)
        if not is_full:
            chars = chars[: len(chunk) + 1]  # n bytes -> n+1 chars
        out.append(chars)
    return "".join(out)


def _decode85(s: str) -> Optional[bytes]:
    out = bytearray()
    group: List[int] = []
    for c in s:
        if c == "z":
            # 'z' is only valid at a group boundary (an empty accumulator).
            if group:
                return None
            out.extend(b"\x00\x00\x00\x00")
            continue
        code = ord(c)
        if code < 33 or code > 117:
            return None
        group.append(code - 33)
        if len(group) == 5:
            v = 0
            for d in group:
                v = v * 85 + d
            if v > 0xFFFFFFFF:
                return None  # a 5-char group must fit in 32 bits
            out.extend(v.to_bytes(4, "big"))
            group = []
    # Handle a partial final group (2-4 chars -> 1-3 bytes).
    if group:
        m = len(group)
        if m < 2:
            return None  # a lone trailing char is malformed
        while len(group) < 5:
            group.append(84)  # pad with 'u'
        v = 0
        for d in group:
            v = v * 85 + d
        if v > 0xFFFFFFFF:
            return None
        all_bytes = v.to_bytes(4, "big")
        out.extend(all_bytes[: m - 1])
    return bytes(out)


# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------


def _encode_bytes(data: bytes, scheme: Scheme) -> str:
    if scheme == "base32":
        return _encode32(data)
    if scheme == "base58":
        return _encode58(data)
    if scheme == "base62":
        return _encode62(data)
    if scheme == "base85":
        return _encode85(data)
    return ""


def _decode_bytes(encoded: str, scheme: Scheme) -> Optional[bytes]:
    if scheme == "base32":
        return _decode32(encoded)
    if scheme == "base58":
        return _decode58(encoded)
    if scheme == "base62":
        return _decode62(encoded)
    if scheme == "base85":
        return _decode85(encoded)
    return None


def encode(text: str, scheme: Scheme) -> str:
    """Encode the UTF-8 bytes of ``text`` per ``scheme``. Empty text -> ``''``.

    Mirrors ``Encode`` in cli/base-encoder/base-encoder.go.
    """
    return _encode_bytes(text.encode("utf-8"), scheme)


def decode(encoded: str, scheme: Scheme) -> Optional[str]:
    """Decode ``encoded`` back to UTF-8 text. Invalid chars / malformed ->
    ``None`` (mirrors the Go twin's ``errInvalid`` and the TS lib's ``null``).

    Mirrors ``Decode`` in cli/base-encoder/base-encoder.go.
    """
    data = _decode_bytes(encoded, scheme)
    if data is None:
        return None
    # Lossy so a structurally-valid-but-non-UTF-8 payload never raises a second
    # error (mirrors Go's string(data), which never fails).
    return data.decode("utf-8", "replace")


# ---------------------------------------------------------------------------
# Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
# Run directly: `python3 python.py`
# ---------------------------------------------------------------------------
if __name__ == "__main__":
    # Base32 — known values + RFC 4648 padding + case sensitivity.
    assert encode("hello", "base32") == "NBSWY3DP"
    assert encode("foo", "base32") == "MZXW6==="  # 3 bytes -> 5 chars + 3 '='
    assert decode("NBSWY3DP", "base32") == "hello"
    assert decode("nbswy3dp", "base32") is None  # lowercase not in RFC 4648

    # Base58 — each leading 0x00 byte -> a leading '1'.
    assert encode("\x00", "base58") == "1"
    assert encode("\x00\x00A", "base58").startswith("11")
    assert decode("1", "base58") == "\x00"
    assert decode(encode("\x00\x00A", "base58"), "base58") == "\x00\x00A"

    # Base62 — plain big-int base conversion (no leading-zero preservation).
    assert encode("A", "base62") == "13"  # 1*62 + 3
    assert decode("13", "base62") == "A"
    assert encode("\x00", "base62") == "0"
    assert decode("0", "base62") == ""  # minimal rep of 0 is empty

    # Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection.
    assert encode("hello", "base85") == "BOu!rDZ"
    assert encode("\x00\x00\x00\x00", "base85") == "z"
    assert encode("\x00" * 8, "base85") == "zz"
    assert decode("uuuuu", "base85") is None  # 5-char group overflows 32 bits
    assert decode("B", "base85") is None  # lone trailing char is malformed

    # Cross-scheme — empty, multibyte round-trip, and invalid rejection.
    for scheme in ("base32", "base58", "base62", "base85"):
        assert encode("", scheme) == ""
        assert decode("", scheme) == ""
        assert decode(encode("CosmoDev \U0001f680", scheme), scheme) == "CosmoDev \U0001f680"
        assert decode("~!not-valid!~", scheme) is None  # '~' outside every alphabet

    print("ok")

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →