Skip to content

Hex ↔ Text Converter — Python source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""hex-converter — pure hex ↔ text conversion.

Language: Python.

CosmoDev polyglot showcase port of the ``hex-converter`` tool.
Ported from src/lib/hexText.ts (the canonical TypeScript implementation).

Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
Deterministic, side-effect free; invalid byte sequences decode to U+FFFD,
matching the canonical logic.
"""

import re
from dataclasses import dataclass
from typing import List, Optional

# How encoded bytes are joined when rendered as a hex string.
DELIM_NONE = "none"
DELIM_SPACE = "space"
DELIM_0X = "0x"
DELIM_BACKSLASH_X = "backslash-x"

# U+FFFD, substituted for malformed UTF-8 on decode.
_REPLACEMENT_CHAR = "�"

# Precompiled patterns mirroring the canonical TS regexes. Python's ``re`` is
# Unicode-aware for ``str`` patterns, so ``\s`` matches the same Unicode
# whitespace class the JS source does.
_MARKER_0X = re.compile(r"0x", re.IGNORECASE)
_MARKER_SLASH_X = re.compile(r"\\x", re.IGNORECASE)  # literal backslash + x
_SEPARATORS = re.compile(r"[\s,:]")
_HEX_ONLY = re.compile(r"^[0-9a-f]+$")


@dataclass
class DecodeResult:
    """Outcome of decoding hex back to text.

    Mirrors the canonical TS surface (ok / text / error) so the shape is
    identical across every language in the polyglot showcase.
    """

    ok: bool
    text: str
    error: Optional[str]

    @classmethod
    def ok_text(cls, text: str) -> "DecodeResult":
        return cls(ok=True, text=text, error=None)


def utf8_encode(text: str) -> List[int]:
    """UTF-8 encode a Python string into a list of byte values (0..255).

    Hand-rolled for byte-exact parity across every showcase language. Python
    strings iterate by code point natively, so astral characters encode as
    4-byte sequences.
    """
    bytes_out: List[int] = []
    for ch in text:
        cp = ord(ch)
        if cp <= 0x7F:
            bytes_out.append(cp)
        elif cp <= 0x7FF:
            bytes_out.append(0xC0 | (cp >> 6))
            bytes_out.append(0x80 | (cp & 0x3F))
        elif cp <= 0xFFFF:
            bytes_out.append(0xE0 | (cp >> 12))
            bytes_out.append(0x80 | ((cp >> 6) & 0x3F))
            bytes_out.append(0x80 | (cp & 0x3F))
        else:
            bytes_out.append(0xF0 | (cp >> 18))
            bytes_out.append(0x80 | ((cp >> 12) & 0x3F))
            bytes_out.append(0x80 | ((cp >> 6) & 0x3F))
            bytes_out.append(0x80 | (cp & 0x3F))
    return bytes_out


def _char_from_code_point(cp: int) -> str:
    """Render a single code point as a string, substituting U+FFFD for any
    value that is not a valid Unicode scalar (surrogates or out of range).

    The canonical TS uses ``String.fromCodePoint`` here, which raises on such
    values; substituting instead keeps the decoder total and consistent with
    its stated "invalid → U+FFFD" contract.
    """
    if 0 <= cp <= 0x10FFFF and not (0xD800 <= cp <= 0xDFFF):
        return chr(cp)
    return _REPLACEMENT_CHAR


def utf8_decode(bytes_in: List[int]) -> str:
    """UTF-8 decode a list of bytes into a string.

    Truncated or invalid sequences yield U+FFFD; missing continuation bytes
    default to 0, mirroring the canonical decoder's lenient reads.
    """
    out: List[str] = []
    i = 0
    n = len(bytes_in)

    def next_byte() -> int:
        nonlocal i
        if i >= n:
            return 0
        b = bytes_in[i]
        i += 1
        return b

    while i < n:
        b = bytes_in[i]
        i += 1
        if b <= 0x7F:
            cp = b
        elif (b >> 5) == 0b110:
            b1 = next_byte()
            cp = ((b & 0x1F) << 6) | (b1 & 0x3F)
        elif (b >> 4) == 0b1110:
            b1 = next_byte()
            b2 = next_byte()
            cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F)
        elif (b >> 3) == 0b11110:
            b1 = next_byte()
            b2 = next_byte()
            b3 = next_byte()
            cp = (
                ((b & 0x07) << 18)
                | ((b1 & 0x3F) << 12)
                | ((b2 & 0x3F) << 6)
                | (b3 & 0x3F)
            )
        else:
            cp = 0xFFFD
        out.append(_char_from_code_point(cp))
    return "".join(out)


def text_to_hex(text: str, delimiter: str = DELIM_NONE, uppercase: bool = False) -> str:
    """Render text as a hex string.

    ``delimiter`` controls how per-byte hex pairs are joined:
      - 'none'         -> "48656c6c6f"
      - 'space'        -> "48 65 6c 6c 6f"
      - '0x'           -> "0x48 0x65 ..."
      - 'backslash-x'  -> "\\x48\\x65..." (no separators, C-style)
    """
    hexes = [f"{b:02x}" for b in utf8_encode(text)]
    if uppercase:
        hexes = [h.upper() for h in hexes]
    if delimiter == DELIM_NONE:
        return "".join(hexes)
    if delimiter == DELIM_SPACE:
        return " ".join(hexes)
    if delimiter == DELIM_0X:
        return " ".join(f"0x{h}" for h in hexes)
    if delimiter == DELIM_BACKSLASH_X:
        return "".join(f"\\x{h}" for h in hexes)
    # Unknown delimiters fall back to no delimiter.
    return "".join(hexes)


def sanitize_hex(text: str) -> str:
    """Strip common affixes users paste alongside hex, then lowercase.

    Removes ``0x`` and ``\\x`` literals (case-insensitive, anywhere),
    whitespace, commas, and colons (MAC-style "aa:bb:cc").
    """
    no_markers = _MARKER_SLASH_X.sub("", _MARKER_0X.sub("", text or ""))
    no_separators = _SEPARATORS.sub("", no_markers)
    return no_separators.lower()


def hex_to_text(hex_str: str, _delimiter: str = DELIM_NONE) -> DecodeResult:
    """Decode a (possibly decorated) hex string back to text.

    Invalid characters and odd lengths are reported via ``error``; valid input
    that contains malformed UTF-8 still decodes with U+FFFD substitution.
    """
    cleaned = sanitize_hex(hex_str)
    if cleaned == "":
        return DecodeResult.ok_text("")
    if _HEX_ONLY.match(cleaned) is None:
        return DecodeResult(False, "", "Hex strings may only contain 0-9 and a-f.")
    if len(cleaned) % 2 != 0:
        return DecodeResult(False, "", "Hex must have an even number of digits.")
    bytes_out = [int(cleaned[i:i + 2], 16) for i in range(0, len(cleaned), 2)]
    return DecodeResult.ok_text(utf8_decode(bytes_out))

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →