Hex ↔ Text Converter — PHP source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* hex-converter — pure hex ↔ text conversion.
*
* Language: PHP.
*
* CosmoDev polyglot showcase port of the `hex-converter` tool.
* Ported from src/lib/hexText.ts (the canonical TypeScript implementation).
*
* Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
* Deterministic, side-effect free; invalid byte sequences decode to U+FFFD,
* matching the canonical logic.
*/
/*
* Result of decoding hex back to text, returned by hex_to_text():
*
* ['ok' => bool, 'text' => string, 'error' => string|null]
*
* Mirrors the canonical TS surface so the shape is identical across every
* language in the polyglot showcase.
*/
/** U+FFFD, substituted for malformed UTF-8 on decode. */
const HEX_REPLACEMENT_CHAR = 0xfffd;
/**
* Encode a single Unicode code point as its UTF-8 byte string.
*
* Centralizes the encoding used by both the encoder and the decoder so the
* two stay in lock-step. Pure ord/chr — no mbstring extension required.
*/
function encode_code_point(int $cp): string {
if ($cp <= 0x7f) {
return chr($cp);
}
if ($cp <= 0x7ff) {
return chr(0xc0 | ($cp >> 6)) . chr(0x80 | ($cp & 0x3f));
}
if ($cp <= 0xffff) {
return chr(0xe0 | ($cp >> 12))
. chr(0x80 | (($cp >> 6) & 0x3f))
. chr(0x80 | ($cp & 0x3f));
}
return chr(0xf0 | ($cp >> 18))
. chr(0x80 | (($cp >> 12) & 0x3f))
. chr(0x80 | (($cp >> 6) & 0x3f))
. chr(0x80 | ($cp & 0x3f));
}
/**
* Return the Unicode code point of a single UTF-8 character (1-4 bytes).
*
* Decodes the character's own bytes directly, avoiding the mbstring extension.
* PCRE's UTF-8 mode (used by utf8_encode_str) guarantees well-formed input.
*/
function code_point_of(string $ch): int {
$n = strlen($ch);
$b0 = ord($ch[0]);
if ($n === 1) {
return $b0;
}
$b1 = ord($ch[1]);
if ($n === 2) {
return (($b0 & 0x1f) << 6) | ($b1 & 0x3f);
}
$b2 = ord($ch[2]);
if ($n === 3) {
return (($b0 & 0x0f) << 12) | (($b1 & 0x3f) << 6) | ($b2 & 0x3f);
}
$b3 = ord($ch[3]);
return (($b0 & 0x07) << 18) | (($b1 & 0x3f) << 12) | (($b2 & 0x3f) << 6) | ($b3 & 0x3f);
}
/**
* Render a code point as a character, substituting U+FFFD for any value that
* is not a valid Unicode scalar (surrogates or out of range). Keeps the
* decoder total / non-throwing on malformed input.
*/
function char_from_code_point(int $cp): string {
$is_surrogate = $cp >= 0xd800 && $cp <= 0xdfff;
$in_range = $cp >= 0 && $cp <= 0x10ffff;
return ($in_range && !$is_surrogate)
? encode_code_point($cp)
: encode_code_point(HEX_REPLACEMENT_CHAR);
}
/**
* UTF-8 encode a string into an array of byte values (0..255).
*
* Hand-rolled for byte-exact parity across every showcase language. Iterates
* by Unicode code point (PCRE's UTF-8 split) so astral characters encode as
* 4-byte sequences.
*/
function utf8_encode_str(string $str): array {
$bytes = [];
// preg_split in UTF-8 mode yields one element per code point (PCRE is
// bundled with PHP — no extension needed). On invalid UTF-8 it returns
// false, which degrades to an empty iteration.
$chars = preg_split('//u', $str, -1, PREG_SPLIT_NO_EMPTY) ?: [];
foreach ($chars as $ch) {
// unpack('C*', ...) yields the byte values of the UTF-8 encoding.
foreach (unpack('C*', encode_code_point(code_point_of($ch))) as $b) {
$bytes[] = $b;
}
}
return $bytes;
}
/**
* UTF-8 decode an array of bytes into a string. Truncated or invalid
* sequences yield U+FFFD; missing continuation bytes default to 0, mirroring
* the canonical decoder's lenient reads.
*/
function utf8_decode_bytes(array $bytes): string {
$out = '';
$i = 0;
$n = count($bytes);
while ($i < $n) {
$b = $bytes[$i++];
if ($b <= 0x7f) {
$cp = $b;
} elseif (($b >> 5) === 0b110) {
$b1 = $i < $n ? $bytes[$i++] : 0;
$cp = (($b & 0x1f) << 6) | ($b1 & 0x3f);
} elseif (($b >> 4) === 0b1110) {
$b1 = $i < $n ? $bytes[$i++] : 0;
$b2 = $i < $n ? $bytes[$i++] : 0;
$cp = (($b & 0x0f) << 12) | (($b1 & 0x3f) << 6) | ($b2 & 0x3f);
} elseif (($b >> 3) === 0b11110) {
$b1 = $i < $n ? $bytes[$i++] : 0;
$b2 = $i < $n ? $bytes[$i++] : 0;
$b3 = $i < $n ? $bytes[$i++] : 0;
$cp = (($b & 0x07) << 18) | (($b1 & 0x3f) << 12) | (($b2 & 0x3f) << 6) | ($b3 & 0x3f);
} else {
$cp = HEX_REPLACEMENT_CHAR;
}
$out .= char_from_code_point($cp);
}
return $out;
}
/**
* Render text as a hex string.
*
* $delimiter controls joining:
* - 'none' → "48656c6c6f"
* - 'space' → "48 65 6c 6c 6f"
* - '0x' → "0x48 0x65 ..."
* - 'backslash-x' → "\x48\x65..." (no separators)
*/
function text_to_hex(string $text, string $delimiter = 'none', bool $uppercase = false): string {
$hexes = array_map(
fn($b) => str_pad(dechex($b), 2, '0', STR_PAD_LEFT),
utf8_encode_str($text)
);
if ($uppercase) {
$hexes = array_map('strtoupper', $hexes);
}
switch ($delimiter) {
case 'none':
return implode('', $hexes);
case 'space':
return implode(' ', $hexes);
case '0x':
return implode(' ', array_map(fn($h) => '0x' . $h, $hexes));
case 'backslash-x':
return implode('', array_map(fn($h) => '\\x' . $h, $hexes));
default:
// Unknown delimiters behave like "none".
return implode('', $hexes);
}
}
/**
* Normalize a hex string before decoding.
*
* Strips `0x` and `\x` literals (case-insensitive, anywhere), whitespace,
* commas, and colons (MAC-style "aa:bb:cc"), then lowercases.
*/
function sanitize_hex(string $input): string {
if ($input === '') {
return '';
}
// (?i) makes each match case-insensitive. In the second pattern, \\\\x in
// the PHP string becomes the regex \\x → a literal backslash followed by x.
$no_0x = preg_replace('/(?i)0x/', '', $input);
$no_bx = preg_replace('/(?i)\\\\x/', '', $no_0x);
$no_sep = preg_replace('/[\s,:]/', '', $no_bx);
return strtolower($no_sep);
}
/**
* Decode a (possibly decorated) hex string back to text.
*
* Invalid characters and odd lengths are reported via the 'error' field;
* valid input that contains malformed UTF-8 still decodes with U+FFFD
* substitution. $delimiter is accepted for API parity but unused.
*/
function hex_to_text(string $hex, string $delimiter = 'none'): array {
$cleaned = sanitize_hex($hex);
if ($cleaned === '') {
return ['ok' => true, 'text' => '', 'error' => null];
}
if (preg_match('/^[0-9a-f]+$/', $cleaned) !== 1) {
return ['ok' => false, 'text' => '', 'error' => 'Hex strings may only contain 0-9 and a-f.'];
}
if (strlen($cleaned) % 2 !== 0) {
return ['ok' => false, 'text' => '', 'error' => 'Hex must have an even number of digits.'];
}
// cleaned is pure ASCII [0-9a-f] here, so byte length equals char count.
$bytes = [];
for ($i = 0; $i < strlen($cleaned); $i += 2) {
$bytes[] = intval(substr($cleaned, $i, 2), 16);
}
return ['ok' => true, 'text' => utf8_decode_bytes($bytes), 'error' => null];
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →