Hex ↔ Text Converter — Python source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""hex-converter — pure hex ↔ text conversion.
Language: Python.
CosmoDev polyglot showcase port of the ``hex-converter`` tool.
Ported from src/lib/hexText.ts (the canonical TypeScript implementation).
Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
Deterministic, side-effect free; invalid byte sequences decode to U+FFFD,
matching the canonical logic.
"""
import re
from dataclasses import dataclass
from typing import List, Optional
# How encoded bytes are joined when rendered as a hex string.
DELIM_NONE = "none"
DELIM_SPACE = "space"
DELIM_0X = "0x"
DELIM_BACKSLASH_X = "backslash-x"
# U+FFFD, substituted for malformed UTF-8 on decode.
_REPLACEMENT_CHAR = "�"
# Precompiled patterns mirroring the canonical TS regexes. Python's ``re`` is
# Unicode-aware for ``str`` patterns, so ``\s`` matches the same Unicode
# whitespace class the JS source does.
_MARKER_0X = re.compile(r"0x", re.IGNORECASE)
_MARKER_SLASH_X = re.compile(r"\\x", re.IGNORECASE) # literal backslash + x
_SEPARATORS = re.compile(r"[\s,:]")
_HEX_ONLY = re.compile(r"^[0-9a-f]+$")
@dataclass
class DecodeResult:
"""Outcome of decoding hex back to text.
Mirrors the canonical TS surface (ok / text / error) so the shape is
identical across every language in the polyglot showcase.
"""
ok: bool
text: str
error: Optional[str]
@classmethod
def ok_text(cls, text: str) -> "DecodeResult":
return cls(ok=True, text=text, error=None)
def utf8_encode(text: str) -> List[int]:
"""UTF-8 encode a Python string into a list of byte values (0..255).
Hand-rolled for byte-exact parity across every showcase language. Python
strings iterate by code point natively, so astral characters encode as
4-byte sequences.
"""
bytes_out: List[int] = []
for ch in text:
cp = ord(ch)
if cp <= 0x7F:
bytes_out.append(cp)
elif cp <= 0x7FF:
bytes_out.append(0xC0 | (cp >> 6))
bytes_out.append(0x80 | (cp & 0x3F))
elif cp <= 0xFFFF:
bytes_out.append(0xE0 | (cp >> 12))
bytes_out.append(0x80 | ((cp >> 6) & 0x3F))
bytes_out.append(0x80 | (cp & 0x3F))
else:
bytes_out.append(0xF0 | (cp >> 18))
bytes_out.append(0x80 | ((cp >> 12) & 0x3F))
bytes_out.append(0x80 | ((cp >> 6) & 0x3F))
bytes_out.append(0x80 | (cp & 0x3F))
return bytes_out
def _char_from_code_point(cp: int) -> str:
"""Render a single code point as a string, substituting U+FFFD for any
value that is not a valid Unicode scalar (surrogates or out of range).
The canonical TS uses ``String.fromCodePoint`` here, which raises on such
values; substituting instead keeps the decoder total and consistent with
its stated "invalid → U+FFFD" contract.
"""
if 0 <= cp <= 0x10FFFF and not (0xD800 <= cp <= 0xDFFF):
return chr(cp)
return _REPLACEMENT_CHAR
def utf8_decode(bytes_in: List[int]) -> str:
"""UTF-8 decode a list of bytes into a string.
Truncated or invalid sequences yield U+FFFD; missing continuation bytes
default to 0, mirroring the canonical decoder's lenient reads.
"""
out: List[str] = []
i = 0
n = len(bytes_in)
def next_byte() -> int:
nonlocal i
if i >= n:
return 0
b = bytes_in[i]
i += 1
return b
while i < n:
b = bytes_in[i]
i += 1
if b <= 0x7F:
cp = b
elif (b >> 5) == 0b110:
b1 = next_byte()
cp = ((b & 0x1F) << 6) | (b1 & 0x3F)
elif (b >> 4) == 0b1110:
b1 = next_byte()
b2 = next_byte()
cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F)
elif (b >> 3) == 0b11110:
b1 = next_byte()
b2 = next_byte()
b3 = next_byte()
cp = (
((b & 0x07) << 18)
| ((b1 & 0x3F) << 12)
| ((b2 & 0x3F) << 6)
| (b3 & 0x3F)
)
else:
cp = 0xFFFD
out.append(_char_from_code_point(cp))
return "".join(out)
def text_to_hex(text: str, delimiter: str = DELIM_NONE, uppercase: bool = False) -> str:
"""Render text as a hex string.
``delimiter`` controls how per-byte hex pairs are joined:
- 'none' -> "48656c6c6f"
- 'space' -> "48 65 6c 6c 6f"
- '0x' -> "0x48 0x65 ..."
- 'backslash-x' -> "\\x48\\x65..." (no separators, C-style)
"""
hexes = [f"{b:02x}" for b in utf8_encode(text)]
if uppercase:
hexes = [h.upper() for h in hexes]
if delimiter == DELIM_NONE:
return "".join(hexes)
if delimiter == DELIM_SPACE:
return " ".join(hexes)
if delimiter == DELIM_0X:
return " ".join(f"0x{h}" for h in hexes)
if delimiter == DELIM_BACKSLASH_X:
return "".join(f"\\x{h}" for h in hexes)
# Unknown delimiters fall back to no delimiter.
return "".join(hexes)
def sanitize_hex(text: str) -> str:
"""Strip common affixes users paste alongside hex, then lowercase.
Removes ``0x`` and ``\\x`` literals (case-insensitive, anywhere),
whitespace, commas, and colons (MAC-style "aa:bb:cc").
"""
no_markers = _MARKER_SLASH_X.sub("", _MARKER_0X.sub("", text or ""))
no_separators = _SEPARATORS.sub("", no_markers)
return no_separators.lower()
def hex_to_text(hex_str: str, _delimiter: str = DELIM_NONE) -> DecodeResult:
"""Decode a (possibly decorated) hex string back to text.
Invalid characters and odd lengths are reported via ``error``; valid input
that contains malformed UTF-8 still decodes with U+FFFD substitution.
"""
cleaned = sanitize_hex(hex_str)
if cleaned == "":
return DecodeResult.ok_text("")
if _HEX_ONLY.match(cleaned) is None:
return DecodeResult(False, "", "Hex strings may only contain 0-9 and a-f.")
if len(cleaned) % 2 != 0:
return DecodeResult(False, "", "Hex must have an even number of digits.")
bytes_out = [int(cleaned[i:i + 2], 16) for i in range(0, len(cleaned), 2)]
return DecodeResult.ok_text(utf8_decode(bytes_out))
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →