Skip to content

Base32 / Base58 / Base62 / Base85 Encoder — Kotlin source

Encode text to Base32, Base58, Base62, or Ascii85 - or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// base-encoder — Base32 (RFC 4648), Base58 (Bitcoin), Base62, and Base85
// (Ascii85) byte-array encoders, operating on the UTF-8 bytes of the input
// text.
//
// Language: Kotlin (1.9, JVM standard library)
// Source:   CosmoDev polyglot showcase port of the Base Encoder tool, ported
//           from cli/base-encoder/base-encoder.go (the authoritative Go twin).
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws (encode always succeeds, decode
//     returns null for invalid or malformed input, mirroring the TS lib's
//     `null` and the Go twin's `errInvalid`).
//   - Functionally equivalent to the Go twin: same inputs -> same outputs.
//   - Self-contained: JVM stdlib only — no external dependencies.
//
// Arbitrary-precision note: Base58 and Base62 base-convert the whole byte
// array. java.math.BigInteger (the JVM equivalent of Go's math/big and
// Python's int) is arbitrary-precision, so we get the exact same semantics
// for free — no manual bignum code (unlike the dependency-free Rust/C/C++
// siblings).

import java.math.BigInteger

/** One of the four supported byte-array base encodings. Mirrors the Go
 * twin's `Scheme` type and the TS `Scheme` union. */
enum class Scheme { BASE32, BASE58, BASE62, BASE85 }

object BaseEncoder {

    private const val B32_ALPHABET = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567"
    private const val B58_ALPHABET =
        "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz"
    private const val B62_ALPHABET =
        "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"

    /** Data characters emitted by a final (partial) 5-byte chunk before '='
     * padding, per RFC 4648. Index = byte count (0..4). Matches the TS
     * `outLen` table. */
    private val OUT_LEN_32 = intArrayOf(0, 2, 4, 5, 7)

    private val BASE_58 = BigInteger.valueOf(58)
    private val BASE_62 = BigInteger.valueOf(62)

    // -----------------------------------------------------------------------
    // BigInteger helper — minimal big-endian byte output, matching Go's
    // big.Int.Bytes() (and Python's int.to_bytes).
    // -----------------------------------------------------------------------

    /** BigInteger -> minimal big-endian bytes. [BigInteger.toByteArray]
     * returns big-endian two's-complement and may carry an extra sign byte;
     * we drop it so the output is minimal. */
    private fun toBigEndianBytes(num: BigInteger): ByteArray {
        if (num.signum() == 0) return ByteArray(0)
        val twoc = num.toByteArray()
        val from = if (twoc[0] == 0.toByte()) 1 else 0 // drop the sign byte
        return twoc.copyOfRange(from, twoc.size)
    }

    // -----------------------------------------------------------------------
    // Base32 — RFC 4648 alphabet, padded to a multiple of 8 chars with '='.
    // -----------------------------------------------------------------------

    private fun encode32(data: ByteArray): String = buildString {
        var i = 0
        while (i < data.size) {
            val n = minOf(5, data.size - i)
            val b = IntArray(5)
            for (j in 0 until n) b[j] = data[i + j].toInt() and 0xFF
            // Pack 5 bytes (40 bits) into 8 base32 digits (5 bits each, big-endian).
            val digits = intArrayOf(
                (b[0] shr 3) and 0x1F,
                ((b[0] shl 2) or (b[1] shr 6)) and 0x1F,
                (b[1] shr 1) and 0x1F,
                ((b[1] shl 4) or (b[2] shr 4)) and 0x1F,
                ((b[2] shl 1) or (b[3] shr 7)) and 0x1F,
                (b[3] shr 2) and 0x1F,
                ((b[3] shl 3) or (b[4] shr 5)) and 0x1F,
                b[4] and 0x1F,
            )
            val outLen = if (n == 5) 8 else OUT_LEN_32[n]
            for (k in 0 until outLen) append(B32_ALPHABET[digits[k]])
            for (k in outLen until 8) append('=')
            i += 5
        }
    }

    private fun decode32(s: String): ByteArray? {
        val out = mutableListOf<Byte>()
        var buffer = 0
        var bits = 0
        for (c in s) {
            if (c == '=') break // padding marks the end
            val idx = B32_ALPHABET.indexOf(c)
            if (idx < 0) return null
            buffer = (buffer shl 5) or idx
            bits += 5
            if (bits >= 8) {
                bits -= 8
                out.add(((buffer shr bits) and 0xFF).toByte())
                buffer = buffer and ((1 shl bits) - 1) // keep only the leftover bits
            }
        }
        return out.toByteArray()
    }

    // -----------------------------------------------------------------------
    // Base58 — Bitcoin alphabet. Leading 0x00 bytes -> leading '1' (count
    // preserved).
    // -----------------------------------------------------------------------

    private fun encode58(data: ByteArray): String {
        // Count leading zero bytes — each maps to a leading '1'.
        var zeros = 0
        while (zeros < data.size && data[zeros] == 0.toByte()) zeros++
        // Big-endian byte array (skipping the leading zeros) -> BigInteger.
        var num = BigInteger.ZERO
        for (i in zeros until data.size) {
            num = num.shiftLeft(8).or(BigInteger.valueOf((data[i].toInt() and 0xFF).toLong()))
        }
        // Base-convert to 58 digits (collected least-significant first).
        val digits = StringBuilder()
        while (num.signum() > 0) {
            val qr = num.divideAndRemainder(BASE_58)
            num = qr[0]
            digits.append(B58_ALPHABET[qr[1].toInt()])
        }
        return "1".repeat(zeros) + digits.reverse()
    }

    private fun decode58(s: String): ByteArray? {
        // Count leading '1's — each maps to a 0x00 byte.
        var zeros = 0
        while (zeros < s.length && s[zeros] == '1') zeros++
        var num = BigInteger.ZERO
        for (i in zeros until s.length) {
            val idx = B58_ALPHABET.indexOf(s[i])
            if (idx < 0) return null
            num = num.multiply(BASE_58).add(BigInteger.valueOf(idx.toLong()))
        }
        // BigInteger -> minimal big-endian bytes.
        val body = toBigEndianBytes(num)
        return ByteArray(zeros + body.size).also { body.copyInto(it, zeros) }
    }

    // -----------------------------------------------------------------------
    // Base62 — standard base-conversion of the byte array (no leading-zero
    // special-casing beyond the standard big-int).
    // -----------------------------------------------------------------------

    private fun encode62(data: ByteArray): String {
        if (data.isEmpty()) return ""
        var num = BigInteger.ZERO
        for (b in data) num = num.shiftLeft(8).or(BigInteger.valueOf((b.toInt() and 0xFF).toLong()))
        if (num.signum() == 0) return "0"
        val digits = StringBuilder()
        while (num.signum() > 0) {
            val qr = num.divideAndRemainder(BASE_62)
            num = qr[0]
            digits.append(B62_ALPHABET[qr[1].toInt()])
        }
        return digits.reverse().toString()
    }

    private fun decode62(s: String): ByteArray? {
        if (s.isEmpty()) return ByteArray(0)
        var num = BigInteger.ZERO
        for (c in s) {
            val idx = B62_ALPHABET.indexOf(c)
            if (idx < 0) return null
            num = num.multiply(BASE_62).add(BigInteger.valueOf(idx.toLong()))
        }
        return toBigEndianBytes(num)
    }

    // -----------------------------------------------------------------------
    // Base85 — Ascii85. 4 bytes -> 5 chars in '!'(33)..'u'(117); a full
    // 4-zero group is shortened to 'z'. No <~ ~> delimiters. Partial final
    // groups emit one fewer char than (bytes+1) would suggest; decode
    // reverses, padding with 'u' (value 84).
    // -----------------------------------------------------------------------

    private fun encode85(data: ByteArray): String = buildString {
        var i = 0
        while (i < data.size) {
            val n = minOf(4, data.size - i)
            val isFull = n == 4
            val b = LongArray(4)
            for (j in 0 until n) b[j] = (data[i + j].toLong() and 0xFF)
            val u = b[0] * 16777216L + b[1] * 65536L + b[2] * 256L + b[3]
            if (isFull && u == 0L) {
                append('z') // zero-group shorthand
                i += 4
                continue
            }
            val digits = LongArray(5)
            var v = u
            for (k in 4 downTo 0) {
                digits[k] = v % 85
                v /= 85
            }
            val emit = if (isFull) 5 else n + 1 // n bytes -> n+1 chars
            for (k in 0 until emit) append((digits[k] + 33).toInt().toChar())
            i += 4
        }
    }

    private fun decode85(s: String): ByteArray? {
        val out = mutableListOf<Byte>()
        val group = mutableListOf<Int>() // accumulated digit values (0..84)
        for (c in s) {
            if (c == 'z') {
                // 'z' is only valid at a group boundary (an empty accumulator).
                if (group.isNotEmpty()) return null
                out.addAll(listOf(0, 0, 0, 0).map { it.toByte() })
                continue
            }
            if (c.code < 33 || c.code > 117) return null
            group += c.code - 33
            if (group.size == 5) {
                var v = 0L
                for (d in group) v = v * 85 + d
                if (v > 0xFFFFFFFFL) return null // a 5-char group must fit in 32 bits
                out.addAll(byteArrayOf(
                    ((v shr 24) and 0xFF).toByte(),
                    ((v shr 16) and 0xFF).toByte(),
                    ((v shr 8) and 0xFF).toByte(),
                    (v and 0xFF).toByte(),
                ).asList())
                group.clear()
            }
        }
        // Handle a partial final group (2-4 chars -> 1-3 bytes).
        if (group.isNotEmpty()) {
            val m = group.size
            if (m < 2) return null // a lone trailing char is malformed
            while (group.size < 5) group += 84 // pad with 'u'
            var v = 0L
            for (d in group) v = v * 85 + d
            if (v > 0xFFFFFFFFL) return null
            val all = byteArrayOf(
                ((v shr 24) and 0xFF).toByte(),
                ((v shr 16) and 0xFF).toByte(),
                ((v shr 8) and 0xFF).toByte(),
                (v and 0xFF).toByte(),
            )
            for (j in 0 until m - 1) out += all[j]
        }
        return out.toByteArray()
    }

    // -----------------------------------------------------------------------
    // Public API
    // -----------------------------------------------------------------------

    /** Dispatch raw bytes to the chosen scheme's encoder. Mirrors the Go
     * twin's private `encodeBytes`. */
    private fun encodeBytes(data: ByteArray, scheme: Scheme): String = when (scheme) {
        Scheme.BASE32 -> encode32(data)
        Scheme.BASE58 -> encode58(data)
        Scheme.BASE62 -> encode62(data)
        Scheme.BASE85 -> encode85(data)
    }

    /** Dispatch an encoded string to the chosen scheme's decoder. An invalid
     * or malformed input yields null (mirroring the TS `null`). Mirrors the
     * Go twin's private `decodeBytes`. */
    private fun decodeBytes(encoded: String, scheme: Scheme): ByteArray? = when (scheme) {
        Scheme.BASE32 -> decode32(encoded)
        Scheme.BASE58 -> decode58(encoded)
        Scheme.BASE62 -> decode62(encoded)
        Scheme.BASE85 -> decode85(encoded)
    }

    /** Returns the chosen-scheme encoding of the UTF-8 bytes of [text].
     * Empty text encodes to "". It is the Kotlin twin of `Encode` in
     * cli/base-encoder/base-encoder.go. */
    fun encode(text: String, scheme: Scheme): String =
        encodeBytes(text.toByteArray(Charsets.UTF_8), scheme)

    /** Reverses an encoded string back to UTF-8 text. Invalid characters or a
     * malformed structure yield null — mirroring the Go twin's `errInvalid`
     * and the TS lib's `null`. It is the Kotlin twin of `Decode` in
     * cli/base-encoder/base-encoder.go.
     *
     * The decoded bytes are interpreted as UTF-8; the JVM UTF-8 decoder
     * replaces malformed input with U+FFD, so a
     * structurally-valid-but-non-UTF-8 payload never throws a second error
     * (mirroring Go's `string(data)`, which never fails). */
    fun decode(encoded: String, scheme: Scheme): String? {
        val data = decodeBytes(encoded, scheme) ?: return null
        return String(data, Charsets.UTF_8)
    }
}

// ---------------------------------------------------------------------------
// Showcase self-test — mirrors cli/base-encoder/base-encoder_test.go vectors.
// Run directly: `kotlin kotlin.kt`.
// ---------------------------------------------------------------------------
fun main() {
    val nul = "\u0000"

    // Base32 — known values + RFC 4648 padding + case sensitivity.
    check(BaseEncoder.encode("hello", Scheme.BASE32) == "NBSWY3DP")
    // 3 bytes -> 5 data chars + 3 '=' pads.
    check(BaseEncoder.encode("foo", Scheme.BASE32) == "MZXW6===")
    check(BaseEncoder.decode("NBSWY3DP", Scheme.BASE32) == "hello")
    // lowercase is not in the RFC 4648 alphabet
    check(BaseEncoder.decode("nbswy3dp", Scheme.BASE32) == null)

    // Base58 — each leading 0x00 byte -> a leading '1'.
    check(BaseEncoder.encode(nul, Scheme.BASE58) == "1")
    check(BaseEncoder.encode(nul + nul + "A", Scheme.BASE58).startsWith("11"))
    check(BaseEncoder.decode("1", Scheme.BASE58) == nul)
    // round-trip preserves the leading zero bytes exactly
    check(BaseEncoder.decode(BaseEncoder.encode(nul + nul + "A", Scheme.BASE58), Scheme.BASE58) ==
        nul + nul + "A")

    // Base62 — plain big-int base conversion (no leading-zero preservation).
    check(BaseEncoder.encode("A", Scheme.BASE62) == "13") // 1*62 + 3
    check(BaseEncoder.decode("13", Scheme.BASE62) == "A")
    check(BaseEncoder.encode(nul, Scheme.BASE62) == "0")
    // no leading-zero preservation: the minimal rep of 0 is empty
    check(BaseEncoder.decode("0", Scheme.BASE62) == "")

    // Base85 — Ascii85 'z' shorthand + 32-bit overflow rejection.
    check(BaseEncoder.encode("hello", Scheme.BASE85) == "BOu!rDZ")
    check(BaseEncoder.encode(nul + nul + nul + nul, Scheme.BASE85) == "z")
    check(BaseEncoder.encode(nul.repeat(8), Scheme.BASE85) == "zz")
    // a 5-char group must fit in 32 bits; "uuuuu" overflows
    check(BaseEncoder.decode("uuuuu", Scheme.BASE85) == null)
    // a lone trailing char is a malformed partial group
    check(BaseEncoder.decode("B", Scheme.BASE85) == null)

    // Cross-scheme — empty, multibyte round-trip, and invalid rejection.
    for (scheme in Scheme.entries) {
        check(BaseEncoder.encode("", scheme) == "")
        check(BaseEncoder.decode("", scheme) == "")
        // multibyte UTF-8 round-trips through every scheme
        check(BaseEncoder.decode(BaseEncoder.encode("CosmoDev 🚀", scheme), scheme) == "CosmoDev 🚀")
        // '~' is outside every supported alphabet
        check(BaseEncoder.decode("~!not-valid!~", scheme) == null)
    }

    println("ok")
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →