Skip to content

Hex ↔ Text Converter — Kotlin source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Kotlin implementation — the same logic the interactive tool runs, in a shareable, citable form.

// hex-converter — pure hex ↔ text conversion.
//
// Language: Kotlin 1.9+ (JVM), standard library only
// Source:   CosmoDev polyglot showcase port of the hex-converter tool,
//           ported from src/lib/hexText.ts (the canonical TypeScript
//           implementation).
// License:  display source — part of CosmoDev's polyglot tool pages
//           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//           byte sequences decode to U+FFFD, matching the canonical logic.

/** How encoded bytes are joined when rendered as a hex string. */
enum class Delimiter {
    /** No separator: "48656c6c6f". */
    NONE,

    /** Single space between bytes: "48 65 6c 6c 6f". */
    SPACE,

    /** Each byte prefixed with "0x", space-separated. */
    PREFIX_0X,

    /** Each byte prefixed with "\x", no separator (C-style). */
    BACKSLASH_X,
}

/**
 * Outcome of decoding hex back to text. Mirrors the canonical TS surface:
 * `ok`, `text`, and `error` (null when ok).
 */
data class DecodeResult(val ok: Boolean, val text: String, val error: String?) {
    companion object {
        fun okText(text: String) = DecodeResult(ok = true, text = text, error = null)

        fun fail(message: String) = DecodeResult(ok = false, text = "", error = message)
    }
}

object HexConverter {

    /** U+FFFD, substituted for malformed UTF-8 on decode. */
    private const val REPLACEMENT_CHAR = "�"

    /**
     * UTF-8 encode a string into a list of byte values (0..255).
     *
     * Hand-rolled for byte-exact parity across every showcase language.
     * `codePointAt` joins surrogate pairs, so astral characters encode as
     * 4-byte sequences.
     */
    fun utf8Encode(text: String): List<Int> {
        val bytes = mutableListOf<Int>()
        var i = 0
        while (i < text.length) {
            val cp = text.codePointAt(i)
            i += Character.charCount(cp)
            when {
                cp <= 0x7F -> bytes.add(cp)
                cp <= 0x7FF -> {
                    bytes.add(0xC0 or (cp shr 6))
                    bytes.add(0x80 or (cp and 0x3F))
                }
                cp <= 0xFFFF -> {
                    bytes.add(0xE0 or (cp shr 12))
                    bytes.add(0x80 or ((cp shr 6) and 0x3F))
                    bytes.add(0x80 or (cp and 0x3F))
                }
                else -> {
                    bytes.add(0xF0 or (cp shr 18))
                    bytes.add(0x80 or ((cp shr 12) and 0x3F))
                    bytes.add(0x80 or ((cp shr 6) and 0x3F))
                    bytes.add(0x80 or (cp and 0x3F))
                }
            }
        }
        return bytes
    }

    /**
     * UTF-8 decode a list of bytes into a string. Truncated or invalid
     * sequences yield U+FFFD; missing continuation bytes are taken as 0,
     * matching the canonical decoder's lenient consumption. Surrogate /
     * out-of-range code points are also mapped to U+FFFD so the decoder is
     * total.
     */
    fun utf8Decode(bytes: List<Int>): String {
        val out = StringBuilder()
        var i = 0

        // Reads past the end return 0 — the canonical decoder's behavior.
        fun nextByte(): Int {
            if (i >= bytes.size) return 0
            return bytes[i++]
        }

        while (i < bytes.size) {
            val b = bytes[i++]
            val cp = when {
                b <= 0x7F -> b
                (b shr 5) == 0b110 -> ((b and 0x1F) shl 6) or (nextByte() and 0x3F)
                (b shr 4) == 0b1110 -> {
                    val b1 = nextByte()
                    val b2 = nextByte()
                    ((b and 0x0F) shl 12) or ((b1 and 0x3F) shl 6) or (b2 and 0x3F)
                }
                (b shr 3) == 0b11110 -> {
                    val b1 = nextByte()
                    val b2 = nextByte()
                    val b3 = nextByte()
                    ((b and 0x07) shl 18) or ((b1 and 0x3F) shl 12) or
                        ((b2 and 0x3F) shl 6) or (b3 and 0x3F)
                }
                else -> 0xFFFD
            }
            out.append(charFromCodePoint(cp))
        }
        return out.toString()
    }

    /** Render a single code point as a string, substituting U+FFFD for any
     *  value that is not a valid Unicode scalar (surrogates or out of range).
     *  `Character.toChars` cannot render such values; substituting keeps the
     *  decoder total, consistent with its "invalid → U+FFFD" contract. */
    private fun charFromCodePoint(cp: Int): String =
        if (cp in 0..0x10FFFF && cp !in 0xD800..0xDFFF) {
            String(Character.toChars(cp))
        } else {
            REPLACEMENT_CHAR
        }

    /**
     * Render text as a hex string.
     *
     * `delimiter` controls how per-byte hex pairs are joined:
     *   - [Delimiter.NONE]        -> "48656c6c6f"
     *   - [Delimiter.SPACE]       -> "48 65 6c 6c 6f"
     *   - [Delimiter.PREFIX_0X]   -> "0x48 0x65 ..."
     *   - [Delimiter.BACKSLASH_X] -> "\\x48\\x65..." (no separators, C-style)
     */
    fun textToHex(
        text: String,
        delimiter: Delimiter = Delimiter.NONE,
        uppercase: Boolean = false,
    ): String {
        var hexes = utf8Encode(text).map { "%02x".format(it) }
        if (uppercase) {
            hexes = hexes.map { it.uppercase() }
        }
        return when (delimiter) {
            Delimiter.NONE -> hexes.joinToString("")
            Delimiter.SPACE -> hexes.joinToString(" ")
            Delimiter.PREFIX_0X -> hexes.joinToString(" ") { "0x$it" }
            Delimiter.BACKSLASH_X -> hexes.joinToString("") { "\\x$it" }
        }
    }

    /**
     * Strip common affixes users paste alongside hex — `0x` and `\x` markers
     * (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
     * "aa:bb:cc") — then lowercase. The `(?U)` flag makes `\s` Unicode-aware,
     * matching the canonical TS regex.
     */
    fun sanitizeHex(input: String?): String {
        if (input == null) return ""
        val noMarkers = input
            .replace(Regex("0x", RegexOption.IGNORE_CASE), "")
            .replace(Regex("\\\\x", RegexOption.IGNORE_CASE), "")
        return noMarkers.replace(Regex("(?U)[\\s,:]"), "").lowercase()
    }

    /**
     * Decode a (possibly decorated) hex string back to text. Invalid
     * characters and odd lengths are reported via `error`; valid input
     * containing malformed UTF-8 still decodes with U+FFFD substitution.
     */
    fun hexToText(hexStr: String?): DecodeResult {
        val cleaned = sanitizeHex(hexStr)
        if (cleaned.isEmpty()) return DecodeResult.okText("")
        // After sanitizing + lowercasing, every char must be in [0-9a-f].
        if (!Regex("^[0-9a-f]+$").matches(cleaned)) {
            return DecodeResult.fail("Hex strings may only contain 0-9 and a-f.")
        }
        if (cleaned.length % 2 != 0) {
            return DecodeResult.fail("Hex must have an even number of digits.")
        }
        val bytes = cleaned.chunked(2).map { Integer.parseInt(it, 16) }
        return DecodeResult.okText(utf8Decode(bytes))
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →