Hex ↔ Text Converter — Rust source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! hex-converter — pure hex ↔ text conversion.
//!
//! Language: Rust.
//!
//! CosmoDev polyglot showcase port of the `hex-converter` tool.
//! Ported from src/lib/hexText.ts (the canonical TypeScript implementation).
//!
//! Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//! Deterministic, side-effect free; invalid byte sequences decode to U+FFFD,
//! matching the canonical logic.
/// How encoded bytes are joined when rendered as a hex string.
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum Delimiter {
/// No separator: "48656c6c6f".
None,
/// Single space between bytes.
Space,
/// Each byte prefixed with "0x", space-separated.
Prefix0x,
/// Each byte prefixed with "\x", no separator (C-style).
BackslashX,
}
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `error` (None when ok).
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct DecodeResult {
pub ok: bool,
pub text: String,
pub error: Option<String>,
}
impl DecodeResult {
fn ok_text(text: String) -> Self {
DecodeResult { ok: true, text, error: None }
}
fn fail(message: &str) -> Self {
DecodeResult { ok: false, text: String::new(), error: Some(message.to_string()) }
}
}
/// U+FFFD, substituted for malformed UTF-8 on decode.
const REPLACEMENT_CHAR: char = '\u{FFFD}';
/// UTF-8 encode a Rust string into a vector of bytes.
///
/// Hand-rolled so every language in the polyglot showcase produces
/// byte-identical output. `chars()` yields Unicode scalar values, so astral
/// characters encode as 4-byte sequences.
pub fn utf8_encode(s: &str) -> Vec<u8> {
let mut bytes = Vec::new();
for c in s.chars() {
let cp = c as u32;
if cp <= 0x7f {
bytes.push(cp as u8);
} else if cp <= 0x7ff {
bytes.push(0xc0 | (cp >> 6) as u8);
bytes.push(0x80 | (cp & 0x3f) as u8);
} else if cp <= 0xffff {
bytes.push(0xe0 | (cp >> 12) as u8);
bytes.push(0x80 | ((cp >> 6) & 0x3f) as u8);
bytes.push(0x80 | (cp & 0x3f) as u8);
} else {
bytes.push(0xf0 | (cp >> 18) as u8);
bytes.push(0x80 | ((cp >> 12) & 0x3f) as u8);
bytes.push(0x80 | ((cp >> 6) & 0x3f) as u8);
bytes.push(0x80 | (cp & 0x3f) as u8);
}
}
bytes
}
/// UTF-8 decode a byte slice into a String. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption. `char::from_u32` returns None for
/// surrogates / out-of-range values, which we also map to U+FFFD so the
/// decoder is total.
pub fn utf8_decode(bytes: &[u8]) -> String {
let mut out = String::new();
let mut i = 0;
while i < bytes.len() {
let b = bytes[i];
i += 1;
let cp: u32 = if b <= 0x7f {
b as u32
} else if b >> 5 == 0b110 {
let b1 = next_byte(bytes, &mut i) as u32;
((b as u32 & 0x1f) << 6) | (b1 & 0x3f)
} else if b >> 4 == 0b1110 {
let b1 = next_byte(bytes, &mut i) as u32;
let b2 = next_byte(bytes, &mut i) as u32;
((b as u32 & 0x0f) << 12) | ((b1 & 0x3f) << 6) | (b2 & 0x3f)
} else if b >> 3 == 0b11110 {
let b1 = next_byte(bytes, &mut i) as u32;
let b2 = next_byte(bytes, &mut i) as u32;
let b3 = next_byte(bytes, &mut i) as u32;
((b as u32 & 0x07) << 18)
| ((b1 & 0x3f) << 12)
| ((b2 & 0x3f) << 6)
| (b3 & 0x3f)
} else {
REPLACEMENT_CHAR as u32
};
out.push(char::from_u32(cp).unwrap_or(REPLACEMENT_CHAR));
}
out
}
/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
fn next_byte(bytes: &[u8], i: &mut usize) -> u8 {
if *i >= bytes.len() {
return 0;
}
let b = bytes[*i];
*i += 1;
b
}
/// Render text as a hex string using the given delimiter and case.
pub fn text_to_hex(text: &str, delimiter: Delimiter, uppercase: bool) -> String {
let mut hexes: Vec<String> = utf8_encode(text)
.iter()
.map(|b| format!("{:02x}", b))
.collect();
if uppercase {
for h in hexes.iter_mut() {
*h = h.to_uppercase();
}
}
match delimiter {
Delimiter::None => hexes.join(""),
Delimiter::Space => hexes.join(" "),
Delimiter::Prefix0x => hexes.iter().map(|h| format!("0x{}", h)).collect::<Vec<_>>().join(" "),
Delimiter::BackslashX => hexes.iter().map(|h| format!("\\x{}", h)).collect::<Vec<_>>().join(""),
}
}
/// Strip common affixes users paste alongside hex — `0x` and `\x` literals,
/// whitespace, commas, and colons (MAC-style "aa:bb:cc") — then lowercase.
/// Safe on empty input.
///
/// The two markers are removed case-insensitively wherever they appear; ASCII
/// lowercasing is sufficient because only [0-9a-f] are valid afterward.
pub fn sanitize_hex(input: &str) -> String {
let no_markers = input
.replace("0x", "")
.replace("0X", "")
.replace("\\x", "")
.replace("\\X", "");
let mut cleaned: String = no_markers
.chars()
.filter(|&c| !c.is_whitespace() && c != ',' && c != ':')
.collect();
cleaned.make_ascii_lowercase();
cleaned
}
/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `error`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
pub fn hex_to_text(hex: &str, _delimiter: Delimiter) -> DecodeResult {
let cleaned = sanitize_hex(hex);
if cleaned.is_empty() {
return DecodeResult::ok_text(String::new());
}
// After sanitizing + lowercasing, every char must be in [0-9a-f].
if !cleaned.chars().all(|c| matches!(c, '0'..='9' | 'a'..='f')) {
return DecodeResult::fail("Hex strings may only contain 0-9 and a-f.");
}
if cleaned.len() % 2 != 0 {
return DecodeResult::fail("Hex must have an even number of digits.");
}
// cleaned is pure ASCII here, so byte chunks of 2 map 1:1 to char pairs.
let bytes: Vec<u8> = cleaned
.as_bytes()
.chunks(2)
.map(|pair| hex_digit(pair[0] as char) * 16 + hex_digit(pair[1] as char))
.collect();
DecodeResult::ok_text(utf8_decode(&bytes))
}
/// Map a single validated hex digit to its numeric value. The wildcard arm is
/// unreachable because callers pre-validate the input.
fn hex_digit(c: char) -> u8 {
match c {
'0'..='9' => c as u8 - b'0',
'a'..='f' => c as u8 - b'a' + 10,
_ => 0,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →