Hex ↔ Text Converter — C++ source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// hex-converter — pure hex ↔ text conversion.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the hex-converter tool,
// ported from src/lib/hexText.ts (the canonical TypeScript
// implementation).
// License: display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org). Deterministic, side-effect free; invalid
// byte sequences decode to U+FFFD, matching the canonical logic.
#include <cctype>
#include <cstdint>
#include <string>
#include <vector>
namespace hex_converter {
/// How encoded bytes are joined when rendered as a hex string.
enum class Delimiter {
None, ///< No separator: "48656c6c6f".
Space, ///< Single space between bytes: "48 65 6c 6c 6f".
Prefix0x, ///< Each byte prefixed with "0x", space-separated.
BackslashX, ///< Each byte prefixed with "\x", no separator (C-style).
};
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `error` (empty when ok).
struct DecodeResult {
bool ok;
std::string text;
std::string error;
static DecodeResult okText(std::string text)
{
return DecodeResult{true, std::move(text), std::string{}};
}
static DecodeResult fail(std::string message)
{
return DecodeResult{false, std::string{}, std::move(message)};
}
};
namespace {
/// U+FFFD code point, substituted for malformed UTF-8 on decode.
constexpr std::uint32_t kReplacementCp = 0xFFFD;
/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
std::uint8_t nextByte(const std::vector<std::uint8_t>& bytes, std::size_t& i)
{
if (i >= bytes.size()) return 0;
return bytes[i++];
}
/// Append one code point as UTF-8, substituting U+FFFD for surrogates and
/// out-of-range values so the decoder is total. (std::codecvt is deprecated
/// in C++17, so the encoding is spelled out.)
void appendCodePoint(std::string& out, std::uint32_t cp)
{
if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = kReplacementCp;
if (cp <= 0x7F) {
out.push_back(static_cast<char>(cp));
} else if (cp <= 0x7FF) {
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else if (cp <= 0xFFFF) {
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
} else {
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
}
}
/// Read one UTF-8 code point starting at text[i]. Invalid lead bytes,
/// truncated sequences, and surrogate/out-of-range values yield U+FFFD; the
/// cursor always advances at least one byte.
std::uint32_t nextCodePoint(const std::string& text, std::size_t& i)
{
auto read = [&]() -> std::uint32_t {
if (i >= text.size()) return 0;
return static_cast<std::uint8_t>(text[i++]);
};
const std::uint32_t b = read();
std::uint32_t cp;
if (b <= 0x7F) {
cp = b;
} else if ((b >> 5) == 0b110) {
cp = ((b & 0x1F) << 6) | (read() & 0x3F);
} else if ((b >> 4) == 0b1110) {
cp = ((b & 0x0F) << 12) | ((read() & 0x3F) << 6) | (read() & 0x3F);
} else if ((b >> 3) == 0b11110) {
cp = ((b & 0x07) << 18) | ((read() & 0x3F) << 12) | ((read() & 0x3F) << 6) |
(read() & 0x3F);
} else {
cp = kReplacementCp;
}
if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = kReplacementCp;
return cp;
}
} // namespace
/// UTF-8 encode text into a vector of byte values (0..255).
///
/// std::string holds bytes and a UTF-8 source string is already its own
/// encoding; the code points are re-derived and pushed through the canonical
/// 1..4-byte branches so every language in the polyglot showcase produces
/// byte-identical output.
std::vector<std::uint8_t> utf8Encode(const std::string& text)
{
std::vector<std::uint8_t> bytes;
std::size_t i = 0;
while (i < text.size()) {
const std::uint32_t cp = nextCodePoint(text, i);
if (cp <= 0x7F) {
bytes.push_back(static_cast<std::uint8_t>(cp));
} else if (cp <= 0x7FF) {
bytes.push_back(static_cast<std::uint8_t>(0xC0 | (cp >> 6)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
} else if (cp <= 0xFFFF) {
bytes.push_back(static_cast<std::uint8_t>(0xE0 | (cp >> 12)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 6) & 0x3F)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
} else {
bytes.push_back(static_cast<std::uint8_t>(0xF0 | (cp >> 18)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 12) & 0x3F)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | ((cp >> 6) & 0x3F)));
bytes.push_back(static_cast<std::uint8_t>(0x80 | (cp & 0x3F)));
}
}
return bytes;
}
/// UTF-8 decode a byte vector into a string. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption.
std::string utf8Decode(const std::vector<std::uint8_t>& bytes)
{
std::string out;
std::size_t i = 0;
while (i < bytes.size()) {
const std::uint8_t b = bytes[i++];
std::uint32_t cp;
if (b <= 0x7F) {
cp = b;
} else if ((b >> 5) == 0b110) {
const std::uint32_t b1 = nextByte(bytes, i);
cp = ((b & 0x1F) << 6) | (b1 & 0x3F);
} else if ((b >> 4) == 0b1110) {
const std::uint32_t b1 = nextByte(bytes, i);
const std::uint32_t b2 = nextByte(bytes, i);
cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
} else if ((b >> 3) == 0b11110) {
const std::uint32_t b1 = nextByte(bytes, i);
const std::uint32_t b2 = nextByte(bytes, i);
const std::uint32_t b3 = nextByte(bytes, i);
cp = ((b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
} else {
cp = kReplacementCp;
}
appendCodePoint(out, cp);
}
return out;
}
/// Render text as a hex string.
///
/// `delimiter` controls how per-byte hex pairs are joined:
/// - None -> "48656c6c6f"
/// - Space -> "48 65 6c 6c 6f"
/// - Prefix0x -> "0x48 0x65 ..."
/// - BackslashX -> "\x48\x65..." (no separators, C-style)
std::string textToHex(const std::string& text, Delimiter delimiter, bool uppercase = false)
{
static const char kLower[] = "0123456789abcdef";
static const char kUpper[] = "0123456789ABCDEF";
const char* digits = uppercase ? kUpper : kLower;
std::vector<std::string> hexes;
for (std::uint8_t b : utf8Encode(text)) {
hexes.emplace_back(std::string{digits[b >> 4], digits[b & 0x0F]});
}
const char* separator = (delimiter == Delimiter::Space || delimiter == Delimiter::Prefix0x)
? " "
: "";
const char* prefix = (delimiter == Delimiter::Prefix0x) ? "0x"
: (delimiter == Delimiter::BackslashX) ? "\\x"
: "";
std::string out;
for (std::size_t k = 0; k < hexes.size(); ++k) {
if (k > 0) out += separator;
out += prefix;
out += hexes[k];
}
return out;
}
/// Strip common affixes users paste alongside hex — `0x` and `\x` markers
/// (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
/// "aa:bb:cc") — then lowercase. ASCII lowercasing is sufficient because only
/// [0-9a-f] are valid afterward; std::isspace covers the ASCII whitespace
/// users actually paste (exotic Unicode spaces cannot contribute valid hex).
std::string sanitizeHex(const std::string& input)
{
std::string noMarkers;
noMarkers.reserve(input.size());
for (std::size_t i = 0; i < input.size();) {
const char c = input[i];
if (i + 1 < input.size() && (c == '0' || c == '\\') &&
(input[i + 1] == 'x' || input[i + 1] == 'X')) {
i += 2; // case-insensitive "0x" / "\x" marker
continue;
}
noMarkers.push_back(c);
i += 1;
}
std::string cleaned;
cleaned.reserve(noMarkers.size());
for (char c : noMarkers) {
const auto uc = static_cast<unsigned char>(c);
if (std::isspace(uc) || c == ',' || c == ':') continue;
cleaned.push_back(static_cast<char>(std::tolower(uc)));
}
return cleaned;
}
namespace {
/// Map a single validated hex digit to its numeric value. The fallback arm is
/// unreachable because callers pre-validate the input.
std::uint8_t hexDigit(char c)
{
if (c >= '0' && c <= '9') return static_cast<std::uint8_t>(c - '0');
if (c >= 'a' && c <= 'f') return static_cast<std::uint8_t>(c - 'a' + 10);
return 0;
}
} // namespace
/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `error`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
DecodeResult hexToText(const std::string& hex)
{
const std::string cleaned = sanitizeHex(hex);
if (cleaned.empty()) return DecodeResult::okText("");
// After sanitizing + lowercasing, every char must be in [0-9a-f].
for (char c : cleaned) {
if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f'))) {
return DecodeResult::fail("Hex strings may only contain 0-9 and a-f.");
}
}
if (cleaned.size() % 2 != 0) {
return DecodeResult::fail("Hex must have an even number of digits.");
}
std::vector<std::uint8_t> bytes;
bytes.reserve(cleaned.size() / 2);
for (std::size_t k = 0; k < cleaned.size(); k += 2) {
bytes.push_back(static_cast<std::uint8_t>(hexDigit(cleaned[k]) * 16 +
hexDigit(cleaned[k + 1])));
}
return DecodeResult::okText(utf8Decode(bytes));
}
} // namespace hex_converter
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →