Hex ↔ Text Converter — C source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* hex-converter — pure hex ↔ text conversion.
*
* Language: C (C11, standard library only)
* Source: CosmoDev polyglot showcase port of the hex-converter tool,
* ported from src/lib/hexText.ts (the canonical TypeScript
* implementation).
* License: display source — part of CosmoDev's polyglot tool pages
* (dev.cosmolabs.org). Deterministic, side-effect free; invalid
* byte sequences decode to U+FFFD, matching the canonical logic.
*/
#include <ctype.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
/* How encoded bytes are joined when rendered as a hex string. */
typedef enum {
DELIM_NONE = 0, /* "48656c6c6f" */
DELIM_SPACE, /* "48 65 6c 6c 6f" */
DELIM_PREFIX_0X, /* "0x48 0x65 0x6c 0x6c 0x6f" */
DELIM_BACKSLASH_X /* "\x48\x65\x6c\x6c\x6f" (C-style, no gaps) */
} delimiter_t;
/* U+FFFD in UTF-8 (EF BF BD), substituted for malformed sequences. */
static const uint8_t REPLACEMENT_UTF8[3] = { 0xEF, 0xBF, 0xBD };
/* Outcome of decoding hex back to text. Mirrors the canonical TS surface:
* ok, text, and error (NULL when ok). The caller frees text and error. */
typedef struct {
bool ok;
char *text; /* heap-allocated; empty string on failure */
char *error; /* heap-allocated; NULL when ok */
} decode_result_t;
static decode_result_t ok_result(char *text)
{
decode_result_t r = { true, text, NULL };
return r;
}
static decode_result_t fail_result(const char *message)
{
decode_result_t r = { false, NULL, NULL };
r.text = calloc(1, 1); /* empty string on failure */
r.error = strdup(message != NULL ? message : "");
return r;
}
void decode_result_free(decode_result_t *r)
{
if (r == NULL) return;
free(r->text);
free(r->error);
r->text = NULL;
r->error = NULL;
}
/* ---- small growable byte buffer ---------------------------------------- */
typedef struct {
uint8_t *data;
size_t len;
size_t cap;
} buf_t;
static bool buf_reserve(buf_t *b, size_t extra)
{
if (b->len + extra <= b->cap) return true;
size_t cap = (b->cap != 0) ? b->cap : 16;
while (cap < b->len + extra) cap *= 2;
uint8_t *p = realloc(b->data, cap);
if (p == NULL) return false;
b->data = p;
b->cap = cap;
return true;
}
static bool buf_push(buf_t *b, uint8_t v)
{
if (!buf_reserve(b, 1)) return false;
b->data[b->len++] = v;
return true;
}
static bool buf_push_bytes(buf_t *b, const uint8_t *src, size_t n)
{
if (!buf_reserve(b, n)) return false;
memcpy(b->data + b->len, src, n);
b->len += n;
return true;
}
static bool buf_push_str(buf_t *b, const char *s)
{
return buf_push_bytes(b, (const uint8_t *)s, strlen(s));
}
/* Terminate the buffer so it doubles as a C string; NULL on OOM. */
static char *buf_finish(buf_t *b)
{
if (!buf_push(b, '\0')) {
free(b->data);
return NULL;
}
return (char *)b->data;
}
/* ---- UTF-8 primitives --------------------------------------------------- */
/* Read the next byte, returning 0 past end-of-input (the canonical decoder's
* behavior) and advancing the cursor. */
static uint8_t next_byte(const uint8_t *s, size_t n, size_t *i)
{
if (*i >= n) return 0;
return s[(*i)++];
}
/* Append cp as UTF-8, substituting U+FFFD for surrogates and out-of-range
* values — String.fromCodePoint's failure mode made total, so the decoder
* never rejects. */
static bool push_code_point(buf_t *out, uint32_t cp)
{
if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = 0xFFFD;
if (cp <= 0x7F) {
return buf_push(out, (uint8_t)cp);
}
if (cp <= 0x7FF) {
return buf_push(out, (uint8_t)(0xC0 | (cp >> 6))) &&
buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
}
if (cp <= 0xFFFF) {
return buf_push(out, (uint8_t)(0xE0 | (cp >> 12))) &&
buf_push(out, (uint8_t)(0x80 | ((cp >> 6) & 0x3F))) &&
buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
}
return buf_push(out, (uint8_t)(0xF0 | (cp >> 18))) &&
buf_push(out, (uint8_t)(0x80 | ((cp >> 12) & 0x3F))) &&
buf_push(out, (uint8_t)(0x80 | ((cp >> 6) & 0x3F))) &&
buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
}
/* Read one UTF-8 code point starting at s[*i]. Invalid lead bytes, truncated
* sequences, and surrogate/out-of-range values yield U+FFFD; the cursor
* always advances at least one byte. */
static uint32_t next_code_point(const uint8_t *s, size_t n, size_t *i)
{
uint8_t b = next_byte(s, n, i);
uint32_t cp;
if (b <= 0x7F) {
cp = b;
} else if ((b >> 5) == 0x6) { /* 110xxxxx */
uint8_t b1 = next_byte(s, n, i);
cp = (((uint32_t)b & 0x1F) << 6) | ((uint32_t)b1 & 0x3F);
} else if ((b >> 4) == 0xE) { /* 1110xxxx */
uint8_t b1 = next_byte(s, n, i);
uint8_t b2 = next_byte(s, n, i);
cp = (((uint32_t)b & 0x0F) << 12) | (((uint32_t)b1 & 0x3F) << 6) |
((uint32_t)b2 & 0x3F);
} else if ((b >> 3) == 0x1E) { /* 11110xxx */
uint8_t b1 = next_byte(s, n, i);
uint8_t b2 = next_byte(s, n, i);
uint8_t b3 = next_byte(s, n, i);
cp = (((uint32_t)b & 0x07) << 18) | (((uint32_t)b1 & 0x3F) << 12) |
(((uint32_t)b2 & 0x3F) << 6) | ((uint32_t)b3 & 0x3F);
} else {
cp = 0xFFFD;
}
if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = 0xFFFD;
return cp;
}
/* UTF-8 encode text into a heap buffer of byte values (0..255), NUL
* terminated. The caller frees the result.
*
* C strings are byte sequences and a UTF-8 source string is already its own
* encoding; the code points are re-derived and pushed through the canonical
* 1..4-byte branches so every language in the polyglot showcase produces
* byte-identical output. */
uint8_t *utf8_encode(const char *text, size_t *out_len)
{
if (text == NULL) text = "";
const uint8_t *s = (const uint8_t *)text;
size_t n = strlen(text);
buf_t out = { NULL, 0, 0 };
size_t i = 0;
while (i < n) {
uint32_t cp = next_code_point(s, n, &i);
if (!push_code_point(&out, cp)) {
free(out.data);
return NULL;
}
}
if (out_len != NULL) *out_len = out.len;
char *str = buf_finish(&out);
if (str == NULL && out_len != NULL) *out_len = 0;
return (uint8_t *)str;
}
/* UTF-8 decode a byte buffer into a NUL-terminated heap string. Truncated or
* invalid sequences yield U+FFFD; missing continuation bytes are taken as 0,
* matching the canonical decoder's lenient consumption. The caller frees. */
char *utf8_decode(const uint8_t *bytes, size_t len)
{
buf_t out = { NULL, 0, 0 };
size_t i = 0;
while (i < len) {
uint8_t b = next_byte(bytes, len, &i);
uint32_t cp;
if (b <= 0x7F) {
cp = b;
} else if ((b >> 5) == 0x6) { /* 110xxxxx */
uint8_t b1 = next_byte(bytes, len, &i);
cp = (((uint32_t)b & 0x1F) << 6) | ((uint32_t)b1 & 0x3F);
} else if ((b >> 4) == 0xE) { /* 1110xxxx */
uint8_t b1 = next_byte(bytes, len, &i);
uint8_t b2 = next_byte(bytes, len, &i);
cp = (((uint32_t)b & 0x0F) << 12) | (((uint32_t)b1 & 0x3F) << 6) |
((uint32_t)b2 & 0x3F);
} else if ((b >> 3) == 0x1E) { /* 11110xxx */
uint8_t b1 = next_byte(bytes, len, &i);
uint8_t b2 = next_byte(bytes, len, &i);
uint8_t b3 = next_byte(bytes, len, &i);
cp = (((uint32_t)b & 0x07) << 18) | (((uint32_t)b1 & 0x3F) << 12) |
(((uint32_t)b2 & 0x3F) << 6) | ((uint32_t)b3 & 0x3F);
} else {
cp = 0xFFFD;
}
if (!push_code_point(&out, cp)) {
free(out.data);
return NULL;
}
}
return buf_finish(&out);
}
/* ---- hex rendering ------------------------------------------------------ */
/* Render text as a hex string. delimiter controls how per-byte hex pairs are
* joined:
* DELIM_NONE -> "48656c6c6f"
* DELIM_SPACE -> "48 65 6c 6c 6f"
* DELIM_PREFIX_0X -> "0x48 0x65 ..."
* DELIM_BACKSLASH_X -> "\x48\x65..." (no separators, C-style)
* Unknown delimiters fall back to no delimiter. */
char *text_to_hex(const char *text, delimiter_t delimiter, bool uppercase)
{
size_t nb = 0;
uint8_t *bytes = utf8_encode(text, &nb);
if (bytes == NULL) return NULL;
static const char lower_digits[] = "0123456789abcdef";
static const char upper_digits[] = "0123456789ABCDEF";
const char *digits = uppercase ? upper_digits : lower_digits;
const char *prefix = (delimiter == DELIM_PREFIX_0X) ? "0x"
: (delimiter == DELIM_BACKSLASH_X) ? "\\x"
: NULL;
buf_t out = { NULL, 0, 0 };
for (size_t k = 0; k < nb; k++) {
if (k > 0 && (delimiter == DELIM_SPACE || delimiter == DELIM_PREFIX_0X)) {
if (!buf_push(&out, ' ')) goto oom;
}
if (prefix != NULL && !buf_push_str(&out, prefix)) goto oom;
if (!buf_push(&out, (uint8_t)digits[bytes[k] >> 4])) goto oom;
if (!buf_push(&out, (uint8_t)digits[bytes[k] & 0x0F])) goto oom;
}
free(bytes);
return buf_finish(&out);
oom:
free(bytes);
free(out.data);
return NULL;
}
/* Strip common affixes users paste alongside hex — "0x" and "\x" markers
* (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
* "aa:bb:cc") — then lowercase. ASCII lowercasing is sufficient because only
* [0-9a-f] are valid afterward. ASCII whitespace filtering is exact for the
* characters users actually paste; exotic Unicode spaces cannot contribute
* valid hex anyway. */
char *sanitize_hex(const char *input)
{
if (input == NULL) input = "";
const uint8_t *s = (const uint8_t *)input;
size_t n = strlen(input);
buf_t out = { NULL, 0, 0 };
size_t i = 0;
while (i < n) {
uint8_t c = s[i];
/* Case-insensitive "0x" / "\x" markers consume two bytes. */
if (i + 1 < n && (c == '0' || c == '\\') &&
(s[i + 1] == 'x' || s[i + 1] == 'X')) {
i += 2;
continue;
}
if (c == ',' || c == ':' || isspace((int)c)) {
i += 1;
continue;
}
if (!buf_push(&out, (uint8_t)tolower((int)c))) {
free(out.data);
return NULL;
}
i += 1;
}
return buf_finish(&out);
}
/* Map a single validated hex digit to its numeric value. The fallback arm is
* unreachable because callers pre-validate the input. */
static int hex_digit(uint8_t c)
{
if (c >= '0' && c <= '9') return (int)(c - '0');
if (c >= 'a' && c <= 'f') return (int)(c - 'a') + 10;
return 0;
}
/* Decode a (possibly decorated) hex string back to text. Invalid characters
* and odd lengths are reported via error; valid input containing malformed
* UTF-8 still decodes with U+FFFD substitution. */
decode_result_t hex_to_text(const char *hex_str)
{
char *cleaned = sanitize_hex(hex_str);
if (cleaned == NULL) return fail_result("out of memory");
size_t n = strlen(cleaned);
if (n == 0) {
free(cleaned);
char *empty = calloc(1, 1);
if (empty == NULL) {
free(cleaned);
return fail_result("out of memory");
}
return ok_result(empty);
}
/* After sanitizing + lowercasing, every char must be in [0-9a-f]. */
for (size_t k = 0; k < n; k++) {
uint8_t c = (uint8_t)cleaned[k];
if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f'))) {
free(cleaned);
return fail_result("Hex strings may only contain 0-9 and a-f.");
}
}
if (n % 2 != 0) {
free(cleaned);
return fail_result("Hex must have an even number of digits.");
}
uint8_t *bytes = malloc(n / 2);
if (bytes == NULL) {
free(cleaned);
return fail_result("out of memory");
}
for (size_t k = 0; k < n / 2; k++) {
bytes[k] = (uint8_t)(hex_digit((uint8_t)cleaned[2 * k]) * 16 +
hex_digit((uint8_t)cleaned[2 * k + 1]));
}
free(cleaned);
char *text = utf8_decode(bytes, n / 2);
free(bytes);
if (text == NULL) return fail_result("out of memory");
return ok_result(text);
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →