Skip to content

Hex ↔ Text Converter — C source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * hex-converter — pure hex ↔ text conversion.
 *
 * Language: C (C11, standard library only)
 * Source:   CosmoDev polyglot showcase port of the hex-converter tool,
 *           ported from src/lib/hexText.ts (the canonical TypeScript
 *           implementation).
 * License:  display source — part of CosmoDev's polyglot tool pages
 *           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
 *           byte sequences decode to U+FFFD, matching the canonical logic.
 */

#include <ctype.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>

/* How encoded bytes are joined when rendered as a hex string. */
typedef enum {
    DELIM_NONE = 0,    /* "48656c6c6f"                             */
    DELIM_SPACE,       /* "48 65 6c 6c 6f"                          */
    DELIM_PREFIX_0X,   /* "0x48 0x65 0x6c 0x6c 0x6f"                */
    DELIM_BACKSLASH_X  /* "\x48\x65\x6c\x6c\x6f" (C-style, no gaps) */
} delimiter_t;

/* U+FFFD in UTF-8 (EF BF BD), substituted for malformed sequences. */
static const uint8_t REPLACEMENT_UTF8[3] = { 0xEF, 0xBF, 0xBD };

/* Outcome of decoding hex back to text. Mirrors the canonical TS surface:
 * ok, text, and error (NULL when ok). The caller frees text and error. */
typedef struct {
    bool  ok;
    char *text;  /* heap-allocated; empty string on failure */
    char *error; /* heap-allocated; NULL when ok */
} decode_result_t;

static decode_result_t ok_result(char *text)
{
    decode_result_t r = { true, text, NULL };
    return r;
}

static decode_result_t fail_result(const char *message)
{
    decode_result_t r = { false, NULL, NULL };
    r.text = calloc(1, 1);           /* empty string on failure */
    r.error = strdup(message != NULL ? message : "");
    return r;
}

void decode_result_free(decode_result_t *r)
{
    if (r == NULL) return;
    free(r->text);
    free(r->error);
    r->text = NULL;
    r->error = NULL;
}

/* ---- small growable byte buffer ---------------------------------------- */

typedef struct {
    uint8_t *data;
    size_t   len;
    size_t   cap;
} buf_t;

static bool buf_reserve(buf_t *b, size_t extra)
{
    if (b->len + extra <= b->cap) return true;
    size_t cap = (b->cap != 0) ? b->cap : 16;
    while (cap < b->len + extra) cap *= 2;
    uint8_t *p = realloc(b->data, cap);
    if (p == NULL) return false;
    b->data = p;
    b->cap = cap;
    return true;
}

static bool buf_push(buf_t *b, uint8_t v)
{
    if (!buf_reserve(b, 1)) return false;
    b->data[b->len++] = v;
    return true;
}

static bool buf_push_bytes(buf_t *b, const uint8_t *src, size_t n)
{
    if (!buf_reserve(b, n)) return false;
    memcpy(b->data + b->len, src, n);
    b->len += n;
    return true;
}

static bool buf_push_str(buf_t *b, const char *s)
{
    return buf_push_bytes(b, (const uint8_t *)s, strlen(s));
}

/* Terminate the buffer so it doubles as a C string; NULL on OOM. */
static char *buf_finish(buf_t *b)
{
    if (!buf_push(b, '\0')) {
        free(b->data);
        return NULL;
    }
    return (char *)b->data;
}

/* ---- UTF-8 primitives --------------------------------------------------- */

/* Read the next byte, returning 0 past end-of-input (the canonical decoder's
 * behavior) and advancing the cursor. */
static uint8_t next_byte(const uint8_t *s, size_t n, size_t *i)
{
    if (*i >= n) return 0;
    return s[(*i)++];
}

/* Append cp as UTF-8, substituting U+FFFD for surrogates and out-of-range
 * values — String.fromCodePoint's failure mode made total, so the decoder
 * never rejects. */
static bool push_code_point(buf_t *out, uint32_t cp)
{
    if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = 0xFFFD;
    if (cp <= 0x7F) {
        return buf_push(out, (uint8_t)cp);
    }
    if (cp <= 0x7FF) {
        return buf_push(out, (uint8_t)(0xC0 | (cp >> 6))) &&
               buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
    }
    if (cp <= 0xFFFF) {
        return buf_push(out, (uint8_t)(0xE0 | (cp >> 12))) &&
               buf_push(out, (uint8_t)(0x80 | ((cp >> 6) & 0x3F))) &&
               buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
    }
    return buf_push(out, (uint8_t)(0xF0 | (cp >> 18))) &&
           buf_push(out, (uint8_t)(0x80 | ((cp >> 12) & 0x3F))) &&
           buf_push(out, (uint8_t)(0x80 | ((cp >> 6) & 0x3F))) &&
           buf_push(out, (uint8_t)(0x80 | (cp & 0x3F)));
}

/* Read one UTF-8 code point starting at s[*i]. Invalid lead bytes, truncated
 * sequences, and surrogate/out-of-range values yield U+FFFD; the cursor
 * always advances at least one byte. */
static uint32_t next_code_point(const uint8_t *s, size_t n, size_t *i)
{
    uint8_t b = next_byte(s, n, i);
    uint32_t cp;
    if (b <= 0x7F) {
        cp = b;
    } else if ((b >> 5) == 0x6) { /* 110xxxxx */
        uint8_t b1 = next_byte(s, n, i);
        cp = (((uint32_t)b & 0x1F) << 6) | ((uint32_t)b1 & 0x3F);
    } else if ((b >> 4) == 0xE) { /* 1110xxxx */
        uint8_t b1 = next_byte(s, n, i);
        uint8_t b2 = next_byte(s, n, i);
        cp = (((uint32_t)b & 0x0F) << 12) | (((uint32_t)b1 & 0x3F) << 6) |
             ((uint32_t)b2 & 0x3F);
    } else if ((b >> 3) == 0x1E) { /* 11110xxx */
        uint8_t b1 = next_byte(s, n, i);
        uint8_t b2 = next_byte(s, n, i);
        uint8_t b3 = next_byte(s, n, i);
        cp = (((uint32_t)b & 0x07) << 18) | (((uint32_t)b1 & 0x3F) << 12) |
             (((uint32_t)b2 & 0x3F) << 6) | ((uint32_t)b3 & 0x3F);
    } else {
        cp = 0xFFFD;
    }
    if (cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)) cp = 0xFFFD;
    return cp;
}

/* UTF-8 encode text into a heap buffer of byte values (0..255), NUL
 * terminated. The caller frees the result.
 *
 * C strings are byte sequences and a UTF-8 source string is already its own
 * encoding; the code points are re-derived and pushed through the canonical
 * 1..4-byte branches so every language in the polyglot showcase produces
 * byte-identical output. */
uint8_t *utf8_encode(const char *text, size_t *out_len)
{
    if (text == NULL) text = "";
    const uint8_t *s = (const uint8_t *)text;
    size_t n = strlen(text);
    buf_t out = { NULL, 0, 0 };
    size_t i = 0;
    while (i < n) {
        uint32_t cp = next_code_point(s, n, &i);
        if (!push_code_point(&out, cp)) {
            free(out.data);
            return NULL;
        }
    }
    if (out_len != NULL) *out_len = out.len;
    char *str = buf_finish(&out);
    if (str == NULL && out_len != NULL) *out_len = 0;
    return (uint8_t *)str;
}

/* UTF-8 decode a byte buffer into a NUL-terminated heap string. Truncated or
 * invalid sequences yield U+FFFD; missing continuation bytes are taken as 0,
 * matching the canonical decoder's lenient consumption. The caller frees. */
char *utf8_decode(const uint8_t *bytes, size_t len)
{
    buf_t out = { NULL, 0, 0 };
    size_t i = 0;
    while (i < len) {
        uint8_t b = next_byte(bytes, len, &i);
        uint32_t cp;
        if (b <= 0x7F) {
            cp = b;
        } else if ((b >> 5) == 0x6) { /* 110xxxxx */
            uint8_t b1 = next_byte(bytes, len, &i);
            cp = (((uint32_t)b & 0x1F) << 6) | ((uint32_t)b1 & 0x3F);
        } else if ((b >> 4) == 0xE) { /* 1110xxxx */
            uint8_t b1 = next_byte(bytes, len, &i);
            uint8_t b2 = next_byte(bytes, len, &i);
            cp = (((uint32_t)b & 0x0F) << 12) | (((uint32_t)b1 & 0x3F) << 6) |
                 ((uint32_t)b2 & 0x3F);
        } else if ((b >> 3) == 0x1E) { /* 11110xxx */
            uint8_t b1 = next_byte(bytes, len, &i);
            uint8_t b2 = next_byte(bytes, len, &i);
            uint8_t b3 = next_byte(bytes, len, &i);
            cp = (((uint32_t)b & 0x07) << 18) | (((uint32_t)b1 & 0x3F) << 12) |
                 (((uint32_t)b2 & 0x3F) << 6) | ((uint32_t)b3 & 0x3F);
        } else {
            cp = 0xFFFD;
        }
        if (!push_code_point(&out, cp)) {
            free(out.data);
            return NULL;
        }
    }
    return buf_finish(&out);
}

/* ---- hex rendering ------------------------------------------------------ */

/* Render text as a hex string. delimiter controls how per-byte hex pairs are
 * joined:
 *   DELIM_NONE         -> "48656c6c6f"
 *   DELIM_SPACE        -> "48 65 6c 6c 6f"
 *   DELIM_PREFIX_0X    -> "0x48 0x65 ..."
 *   DELIM_BACKSLASH_X  -> "\x48\x65..." (no separators, C-style)
 * Unknown delimiters fall back to no delimiter. */
char *text_to_hex(const char *text, delimiter_t delimiter, bool uppercase)
{
    size_t nb = 0;
    uint8_t *bytes = utf8_encode(text, &nb);
    if (bytes == NULL) return NULL;

    static const char lower_digits[] = "0123456789abcdef";
    static const char upper_digits[] = "0123456789ABCDEF";
    const char *digits = uppercase ? upper_digits : lower_digits;
    const char *prefix = (delimiter == DELIM_PREFIX_0X)    ? "0x"
                         : (delimiter == DELIM_BACKSLASH_X) ? "\\x"
                                                            : NULL;

    buf_t out = { NULL, 0, 0 };
    for (size_t k = 0; k < nb; k++) {
        if (k > 0 && (delimiter == DELIM_SPACE || delimiter == DELIM_PREFIX_0X)) {
            if (!buf_push(&out, ' ')) goto oom;
        }
        if (prefix != NULL && !buf_push_str(&out, prefix)) goto oom;
        if (!buf_push(&out, (uint8_t)digits[bytes[k] >> 4])) goto oom;
        if (!buf_push(&out, (uint8_t)digits[bytes[k] & 0x0F])) goto oom;
    }
    free(bytes);
    return buf_finish(&out);

oom:
    free(bytes);
    free(out.data);
    return NULL;
}

/* Strip common affixes users paste alongside hex — "0x" and "\x" markers
 * (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
 * "aa:bb:cc") — then lowercase. ASCII lowercasing is sufficient because only
 * [0-9a-f] are valid afterward. ASCII whitespace filtering is exact for the
 * characters users actually paste; exotic Unicode spaces cannot contribute
 * valid hex anyway. */
char *sanitize_hex(const char *input)
{
    if (input == NULL) input = "";
    const uint8_t *s = (const uint8_t *)input;
    size_t n = strlen(input);

    buf_t out = { NULL, 0, 0 };
    size_t i = 0;
    while (i < n) {
        uint8_t c = s[i];
        /* Case-insensitive "0x" / "\x" markers consume two bytes. */
        if (i + 1 < n && (c == '0' || c == '\\') &&
            (s[i + 1] == 'x' || s[i + 1] == 'X')) {
            i += 2;
            continue;
        }
        if (c == ',' || c == ':' || isspace((int)c)) {
            i += 1;
            continue;
        }
        if (!buf_push(&out, (uint8_t)tolower((int)c))) {
            free(out.data);
            return NULL;
        }
        i += 1;
    }
    return buf_finish(&out);
}

/* Map a single validated hex digit to its numeric value. The fallback arm is
 * unreachable because callers pre-validate the input. */
static int hex_digit(uint8_t c)
{
    if (c >= '0' && c <= '9') return (int)(c - '0');
    if (c >= 'a' && c <= 'f') return (int)(c - 'a') + 10;
    return 0;
}

/* Decode a (possibly decorated) hex string back to text. Invalid characters
 * and odd lengths are reported via error; valid input containing malformed
 * UTF-8 still decodes with U+FFFD substitution. */
decode_result_t hex_to_text(const char *hex_str)
{
    char *cleaned = sanitize_hex(hex_str);
    if (cleaned == NULL) return fail_result("out of memory");
    size_t n = strlen(cleaned);

    if (n == 0) {
        free(cleaned);
        char *empty = calloc(1, 1);
        if (empty == NULL) {
            free(cleaned);
            return fail_result("out of memory");
        }
        return ok_result(empty);
    }
    /* After sanitizing + lowercasing, every char must be in [0-9a-f]. */
    for (size_t k = 0; k < n; k++) {
        uint8_t c = (uint8_t)cleaned[k];
        if (!((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f'))) {
            free(cleaned);
            return fail_result("Hex strings may only contain 0-9 and a-f.");
        }
    }
    if (n % 2 != 0) {
        free(cleaned);
        return fail_result("Hex must have an even number of digits.");
    }

    uint8_t *bytes = malloc(n / 2);
    if (bytes == NULL) {
        free(cleaned);
        return fail_result("out of memory");
    }
    for (size_t k = 0; k < n / 2; k++) {
        bytes[k] = (uint8_t)(hex_digit((uint8_t)cleaned[2 * k]) * 16 +
                             hex_digit((uint8_t)cleaned[2 * k + 1]));
    }
    free(cleaned);

    char *text = utf8_decode(bytes, n / 2);
    free(bytes);
    if (text == NULL) return fail_result("out of memory");
    return ok_result(text);
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →