Skip to content

Hex ↔ Text Converter — Rust source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! hex-converter — pure hex ↔ text conversion.
//!
//! Language: Rust.
//!
//! CosmoDev polyglot showcase port of the `hex-converter` tool.
//! Ported from src/lib/hexText.ts (the canonical TypeScript implementation).
//!
//! Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//! Deterministic, side-effect free; invalid byte sequences decode to U+FFFD,
//! matching the canonical logic.

/// How encoded bytes are joined when rendered as a hex string.
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub enum Delimiter {
    /// No separator: "48656c6c6f".
    None,
    /// Single space between bytes.
    Space,
    /// Each byte prefixed with "0x", space-separated.
    Prefix0x,
    /// Each byte prefixed with "\x", no separator (C-style).
    BackslashX,
}

/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `error` (None when ok).
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct DecodeResult {
    pub ok: bool,
    pub text: String,
    pub error: Option<String>,
}

impl DecodeResult {
    fn ok_text(text: String) -> Self {
        DecodeResult { ok: true, text, error: None }
    }

    fn fail(message: &str) -> Self {
        DecodeResult { ok: false, text: String::new(), error: Some(message.to_string()) }
    }
}

/// U+FFFD, substituted for malformed UTF-8 on decode.
const REPLACEMENT_CHAR: char = '\u{FFFD}';

/// UTF-8 encode a Rust string into a vector of bytes.
///
/// Hand-rolled so every language in the polyglot showcase produces
/// byte-identical output. `chars()` yields Unicode scalar values, so astral
/// characters encode as 4-byte sequences.
pub fn utf8_encode(s: &str) -> Vec<u8> {
    let mut bytes = Vec::new();
    for c in s.chars() {
        let cp = c as u32;
        if cp <= 0x7f {
            bytes.push(cp as u8);
        } else if cp <= 0x7ff {
            bytes.push(0xc0 | (cp >> 6) as u8);
            bytes.push(0x80 | (cp & 0x3f) as u8);
        } else if cp <= 0xffff {
            bytes.push(0xe0 | (cp >> 12) as u8);
            bytes.push(0x80 | ((cp >> 6) & 0x3f) as u8);
            bytes.push(0x80 | (cp & 0x3f) as u8);
        } else {
            bytes.push(0xf0 | (cp >> 18) as u8);
            bytes.push(0x80 | ((cp >> 12) & 0x3f) as u8);
            bytes.push(0x80 | ((cp >> 6) & 0x3f) as u8);
            bytes.push(0x80 | (cp & 0x3f) as u8);
        }
    }
    bytes
}

/// UTF-8 decode a byte slice into a String. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption. `char::from_u32` returns None for
/// surrogates / out-of-range values, which we also map to U+FFFD so the
/// decoder is total.
pub fn utf8_decode(bytes: &[u8]) -> String {
    let mut out = String::new();
    let mut i = 0;
    while i < bytes.len() {
        let b = bytes[i];
        i += 1;
        let cp: u32 = if b <= 0x7f {
            b as u32
        } else if b >> 5 == 0b110 {
            let b1 = next_byte(bytes, &mut i) as u32;
            ((b as u32 & 0x1f) << 6) | (b1 & 0x3f)
        } else if b >> 4 == 0b1110 {
            let b1 = next_byte(bytes, &mut i) as u32;
            let b2 = next_byte(bytes, &mut i) as u32;
            ((b as u32 & 0x0f) << 12) | ((b1 & 0x3f) << 6) | (b2 & 0x3f)
        } else if b >> 3 == 0b11110 {
            let b1 = next_byte(bytes, &mut i) as u32;
            let b2 = next_byte(bytes, &mut i) as u32;
            let b3 = next_byte(bytes, &mut i) as u32;
            ((b as u32 & 0x07) << 18)
                | ((b1 & 0x3f) << 12)
                | ((b2 & 0x3f) << 6)
                | (b3 & 0x3f)
        } else {
            REPLACEMENT_CHAR as u32
        };
        out.push(char::from_u32(cp).unwrap_or(REPLACEMENT_CHAR));
    }
    out
}

/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
fn next_byte(bytes: &[u8], i: &mut usize) -> u8 {
    if *i >= bytes.len() {
        return 0;
    }
    let b = bytes[*i];
    *i += 1;
    b
}

/// Render text as a hex string using the given delimiter and case.
pub fn text_to_hex(text: &str, delimiter: Delimiter, uppercase: bool) -> String {
    let mut hexes: Vec<String> = utf8_encode(text)
        .iter()
        .map(|b| format!("{:02x}", b))
        .collect();
    if uppercase {
        for h in hexes.iter_mut() {
            *h = h.to_uppercase();
        }
    }
    match delimiter {
        Delimiter::None => hexes.join(""),
        Delimiter::Space => hexes.join(" "),
        Delimiter::Prefix0x => hexes.iter().map(|h| format!("0x{}", h)).collect::<Vec<_>>().join(" "),
        Delimiter::BackslashX => hexes.iter().map(|h| format!("\\x{}", h)).collect::<Vec<_>>().join(""),
    }
}

/// Strip common affixes users paste alongside hex — `0x` and `\x` literals,
/// whitespace, commas, and colons (MAC-style "aa:bb:cc") — then lowercase.
/// Safe on empty input.
///
/// The two markers are removed case-insensitively wherever they appear; ASCII
/// lowercasing is sufficient because only [0-9a-f] are valid afterward.
pub fn sanitize_hex(input: &str) -> String {
    let no_markers = input
        .replace("0x", "")
        .replace("0X", "")
        .replace("\\x", "")
        .replace("\\X", "");
    let mut cleaned: String = no_markers
        .chars()
        .filter(|&c| !c.is_whitespace() && c != ',' && c != ':')
        .collect();
    cleaned.make_ascii_lowercase();
    cleaned
}

/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `error`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
pub fn hex_to_text(hex: &str, _delimiter: Delimiter) -> DecodeResult {
    let cleaned = sanitize_hex(hex);
    if cleaned.is_empty() {
        return DecodeResult::ok_text(String::new());
    }
    // After sanitizing + lowercasing, every char must be in [0-9a-f].
    if !cleaned.chars().all(|c| matches!(c, '0'..='9' | 'a'..='f')) {
        return DecodeResult::fail("Hex strings may only contain 0-9 and a-f.");
    }
    if cleaned.len() % 2 != 0 {
        return DecodeResult::fail("Hex must have an even number of digits.");
    }
    // cleaned is pure ASCII here, so byte chunks of 2 map 1:1 to char pairs.
    let bytes: Vec<u8> = cleaned
        .as_bytes()
        .chunks(2)
        .map(|pair| hex_digit(pair[0] as char) * 16 + hex_digit(pair[1] as char))
        .collect();
    DecodeResult::ok_text(utf8_decode(&bytes))
}

/// Map a single validated hex digit to its numeric value. The wildcard arm is
/// unreachable because callers pre-validate the input.
fn hex_digit(c: char) -> u8 {
    match c {
        '0'..='9' => c as u8 - b'0',
        'a'..='f' => c as u8 - b'a' + 10,
        _ => 0,
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →