Skip to content

Hex ↔ Text Converter — C++ source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// hex-converter — pure hex ↔ text conversion.
//
// Language: C++ (C++17, standard library only)
// Source:   CosmoDev polyglot showcase port of the hex-converter tool,
//           ported from src/lib/hexText.ts (the canonical TypeScript
//           implementation).
// License:  display source — part of CosmoDev's polyglot tool pages
//           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//           byte sequences decode to U+FFFD, matching the canonical logic.

#include <cctype>
#include <cstdint>
#include <string>
#include <vector>

namespace hex_converter {

/// How encoded bytes are joined when rendered as a hex string.
enum class Delimiter {
    None,       ///< No separator: "48656c6c6f".
    Space,      ///< Single space between bytes: "48 65 6c 6c 6f".
    Prefix0x,   ///< Each byte prefixed with "0x", space-separated.
    BackslashX, ///< Each byte prefixed with "\x", no separator (C-style).
};

/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `error` (empty when ok).
struct DecodeResult {
    bool ok;
    std::string text;
    std::string error;

    static DecodeResult okText(std::string text)
    {
        return DecodeResult{true, std::move(text), std::string{}};
    }

    static DecodeResult fail(std::string message)
    {
        return DecodeResult{false, std::string{}, std::move(message)};
    }
};

namespace {

/// U+FFFD code point, substituted for malformed UTF-8 on decode.
constexpr std::uint32_t kReplacementCp = 0xFFFD;

/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
std::uint8_t nextByte(const std::vector<std::uint8_t>& bytes, std::size_t& i)
{
    if (i >= bytes.size()) return 0;
    return bytes[i++];
}

/// Append one code point as UTF-8, substituting U+FFFD for surrogates and
/// out-of-range values so the decoder is total. (std::codecvt is deprecated
/// in C++17, so the encoding is spelled out.)
void appendCodePoint(std::string& out, std::uint32_t cp)
{
    if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = kReplacementCp;
    if (cp <= 0x7F) {
        out.push_back(static_cast<char>(cp));
    } else if (cp <= 0x7FF) {
        out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
        out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
    } else if (cp <= 0xFFFF) {
        out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
        out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
        out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
    } else {
        out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
        out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
        out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
        out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
    }
}

/// Read one UTF-8 code point starting at text[i]. Invalid lead bytes,
/// truncated sequences, and surrogate/out-of-range values yield U+FFFD; the
/// cursor always advances at least one byte.
std::uint32_t nextCodePoint(const std::string& text, std::size_t& i)
{
    auto read = [&]() -> std::uint32_t {
        if (i >= text.size()) return 0;
        return static_cast<std::uint8_t>(text[i++]);
    };
    const std::uint32_t b = read();
    std::uint32_t cp;
    if (b <= 0x7F) {
        cp = b;
    } else if ((b >> 5) == 0b110) {
        cp = ((b & 0x1F) << 6) | (read() & 0x3F);
    } else if ((b >> 4) == 0b1110) {
        cp = ((b & 0x0F) << 12) | ((read() & 0x3F) << 6) | (read() & 0x3F);
    } else if ((b >> 3) == 0b11110) {
        cp = ((b & 0x07) << 18) | ((read() & 0x3F) << 12) | ((read() & 0x3F) << 6) |
             (read() & 0x3F);
    } else {
        cp = kReplacementCp;
    }
    if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = kReplacementCp;
    return cp;
}

} // namespace

/// UTF-8 encode text into a vector of byte values (0..255).
///
/// std::string holds bytes and a UTF-8 source string is already its own
/// encoding; the code points are re-derived and pushed through the canonical
/// 1..4-byte branches so every language in the polyglot showcase produces
/// byte-identical output.
std::vector<std::uint8_t> utf8Encode(const std::string& text)
{
    std::vector<std::uint8_t> bytes;
    std::size_t i = 0;
    while (i < text.size()) {
        const std::uint32_t cp = nextCodePoint(text, i);
        if (cp <= 0x7F) {
            bytes.push_back(static_cast<std::uint8_t>(cp));
        } else if (cp <= 0x7FF) {
            bytes.push_back(static_cast<std::uint8_t>(0xC0 | (cp >> 6)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
        } else if (cp <= 0xFFFF) {
            bytes.push_back(static_cast<std::uint8_t>(0xE0 | (cp >> 12)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 6) & 0x3F)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
        } else {
            bytes.push_back(static_cast<std::uint8_t>(0xF0 | (cp >> 18)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 12) & 0x3F)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 6) & 0x3F)));
            bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
        }
    }
    return bytes;
}

/// UTF-8 decode a byte vector into a string. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption.
std::string utf8Decode(const std::vector<std::uint8_t>& bytes)
{
    std::string out;
    std::size_t i = 0;
    while (i < bytes.size()) {
        const std::uint8_t b = bytes[i++];
        std::uint32_t cp;
        if (b <= 0x7F) {
            cp = b;
        } else if ((b >> 5) == 0b110) {
            const std::uint32_t b1 = nextByte(bytes, i);
            cp = ((b & 0x1F) << 6) | (b1 & 0x3F);
        } else if ((b >> 4) == 0b1110) {
            const std::uint32_t b1 = nextByte(bytes, i);
            const std::uint32_t b2 = nextByte(bytes, i);
            cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
        } else if ((b >> 3) == 0b11110) {
            const std::uint32_t b1 = nextByte(bytes, i);
            const std::uint32_t b2 = nextByte(bytes, i);
            const std::uint32_t b3 = nextByte(bytes, i);
            cp = ((b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
        } else {
            cp = kReplacementCp;
        }
        appendCodePoint(out, cp);
    }
    return out;
}

/// Render text as a hex string.
///
/// `delimiter` controls how per-byte hex pairs are joined:
///   - None       -> "48656c6c6f"
///   - Space      -> "48 65 6c 6c 6f"
///   - Prefix0x   -> "0x48 0x65 ..."
///   - BackslashX -> "\x48\x65..." (no separators, C-style)
std::string textToHex(const std::string& text, Delimiter delimiter, bool uppercase = false)
{
    static const char kLower[] = "0123456789abcdef";
    static const char kUpper[] = "0123456789ABCDEF";
    const char* digits = uppercase ? kUpper : kLower;

    std::vector<std::string> hexes;
    for (std::uint8_t b : utf8Encode(text)) {
        hexes.emplace_back(std::string{digits[b >> 4], digits[b & 0x0F]});
    }

    const char* separator = (delimiter == Delimiter::Space || delimiter == Delimiter::Prefix0x)
                                ? " "
                                : "";
    const char* prefix = (delimiter == Delimiter::Prefix0x)    ? "0x"
                         : (delimiter == Delimiter::BackslashX) ? "\\x"
                                                                : "";
    std::string out;
    for (std::size_t k = 0; k < hexes.size(); ++k) {
        if (k > 0) out += separator;
        out += prefix;
        out += hexes[k];
    }
    return out;
}

/// Strip common affixes users paste alongside hex — `0x` and `\x` markers
/// (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
/// "aa:bb:cc") — then lowercase. ASCII lowercasing is sufficient because only
/// [0-9a-f] are valid afterward; std::isspace covers the ASCII whitespace
/// users actually paste (exotic Unicode spaces cannot contribute valid hex).
std::string sanitizeHex(const std::string& input)
{
    std::string noMarkers;
    noMarkers.reserve(input.size());
    for (std::size_t i = 0; i < input.size();) {
        const char c = input[i];
        if (i + 1 < input.size() && (c == '0' || c == '\\') &&
            (input[i + 1] == 'x' || input[i + 1] == 'X')) {
            i += 2; // case-insensitive "0x" / "\x" marker
            continue;
        }
        noMarkers.push_back(c);
        i += 1;
    }

    std::string cleaned;
    cleaned.reserve(noMarkers.size());
    for (char c : noMarkers) {
        const auto uc = static_cast<unsigned char>(c);
        if (std::isspace(uc) || c == ',' || c == ':') continue;
        cleaned.push_back(static_cast<char>(std::tolower(uc)));
    }
    return cleaned;
}

namespace {

/// Map a single validated hex digit to its numeric value. The fallback arm is
/// unreachable because callers pre-validate the input.
std::uint8_t hexDigit(char c)
{
    if (c >= '0' && c <= '9') return static_cast<std::uint8_t>(c - '0');
    if (c >= 'a' && c <= 'f') return static_cast<std::uint8_t>(c - 'a' + 10);
    return 0;
}

} // namespace

/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `error`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
DecodeResult hexToText(const std::string& hex)
{
    const std::string cleaned = sanitizeHex(hex);
    if (cleaned.empty()) return DecodeResult::okText("");

    // After sanitizing + lowercasing, every char must be in [0-9a-f].
    for (char c : cleaned) {
        if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f'))) {
            return DecodeResult::fail("Hex strings may only contain 0-9 and a-f.");
        }
    }
    if (cleaned.size() % 2 != 0) {
        return DecodeResult::fail("Hex must have an even number of digits.");
    }

    std::vector<std::uint8_t> bytes;
    bytes.reserve(cleaned.size() / 2);
    for (std::size_t k = 0; k < cleaned.size(); k += 2) {
        bytes.push_back(static_cast<std::uint8_t>(hexDigit(cleaned[k]) * 16 +
                                                  hexDigit(cleaned[k + 1])));
    }
    return DecodeResult::okText(utf8Decode(bytes));
}

} // namespace hex_converter

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →