Skip to content

Hex ↔ Text Converter — Zig source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! hex-converter — pure hex ↔ text conversion.
//!
//! Language: Zig 0.13 (standard library only)
//! Source:   CosmoDev polyglot showcase port of the hex-converter tool,
//!           ported from src/lib/hexText.ts (the canonical TypeScript
//!           implementation).
//! License:  display source — part of CosmoDev's polyglot tool pages
//!           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//!           byte sequences decode to U+FFFD, matching the canonical logic.

const std = @import("std");

/// How encoded bytes are joined when rendered as a hex string.
pub const Delimiter = enum {
    /// No separator: "48656c6c6f".
    none,
    /// Single space between bytes: "48 65 6c 6c 6f".
    space,
    /// Each byte prefixed with "0x", space-separated.
    prefix_0x,
    /// Each byte prefixed with "\x", no separator (C-style).
    backslash_x,
};

/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `err` (null when ok; `error` is a Zig keyword).
///
/// On success `text` is allocated with the caller's allocator and owned by
/// the caller (empty input yields a static empty literal — freeing a
/// zero-length slice is a no-op in Zig, so `allocator.free(result.text)` is
/// always safe).
pub const DecodeResult = struct {
    ok: bool,
    text: []const u8,
    err: ?[]const u8,

    pub fn okText(text: []const u8) DecodeResult {
        return .{ .ok = true, .text = text, .err = null };
    }

    pub fn fail(message: []const u8) DecodeResult {
        return .{ .ok = false, .text = "", .err = message };
    }
};

/// U+FFFD in UTF-8 (EF BF BD), substituted for malformed sequences.
const replacement_utf8 = [3]u8{ 0xEF, 0xBF, 0xBD };

/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
fn nextByte(bytes: []const u8, i: *usize) u8 {
    if (i.* >= bytes.len) return 0;
    const b = bytes[i.*];
    i.* += 1;
    return b;
}

/// UTF-8 encode a string into bytes.
///
/// Zig string literals are already UTF-8; the code points are re-derived and
/// pushed through the canonical 1..4-byte branches below so every language in
/// the polyglot showcase produces byte-identical output. Input is required to
/// be valid UTF-8 (Zig string literals always are).
pub fn utf8Encode(allocator: std.mem.Allocator, text: []const u8) ![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();

    const view = std.unicode.Utf8View.initUnchecked(text);
    var it = view.iterator();
    while (it.nextCodePoint()) |cp| {
        if (cp <= 0x7F) {
            try out.append(@intCast(cp));
        } else if (cp <= 0x7FF) {
            try out.append(@intCast(0xC0 | (cp >> 6)));
            try out.append(@intCast(0x80 | (cp & 0x3F)));
        } else if (cp <= 0xFFFF) {
            try out.append(@intCast(0xE0 | (cp >> 12)));
            try out.append(@intCast(0x80 | ((cp >> 6) & 0x3F)));
            try out.append(@intCast(0x80 | (cp & 0x3F)));
        } else {
            try out.append(@intCast(0xF0 | (cp >> 18)));
            try out.append(@intCast(0x80 | ((cp >> 12) & 0x3F)));
            try out.append(@intCast(0x80 | ((cp >> 6) & 0x3F)));
            try out.append(@intCast(0x80 | (cp & 0x3F)));
        }
    }
    return out.toOwnedSlice();
}

/// Append one code point as UTF-8, substituting U+FFFD for surrogates and
/// out-of-range values — `utf8Encode`'s error path made total, so the decoder
/// never rejects.
fn appendCodePoint(out: *std.ArrayList(u8), cp: u32) !void {
    if (cp > 0x10FFFF or (cp >= 0xD800 and cp <= 0xDFFF)) {
        try out.appendSlice(&replacement_utf8);
        return;
    }
    var buf: [4]u8 = undefined;
    const n = try std.unicode.utf8Encode(@intCast(cp), &buf);
    try out.appendSlice(buf[0..n]);
}

/// UTF-8 decode a byte slice into a string (UTF-8). Truncated or invalid
/// sequences yield U+FFFD; missing continuation bytes are taken as 0,
/// matching the canonical decoder's lenient consumption.
pub fn utf8Decode(allocator: std.mem.Allocator, bytes: []const u8) ![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();

    var i: usize = 0;
    while (i < bytes.len) {
        const b: u32 = bytes[i];
        i += 1;
        var cp: u32 = undefined;
        if (b <= 0x7F) {
            cp = b;
        } else if (b >> 5 == 0b110) {
            const b1: u32 = nextByte(bytes, &i);
            cp = ((b & 0x1F) << 6) | (b1 & 0x3F);
        } else if (b >> 4 == 0b1110) {
            const b1: u32 = nextByte(bytes, &i);
            const b2: u32 = nextByte(bytes, &i);
            cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
        } else if (b >> 3 == 0b11110) {
            const b1: u32 = nextByte(bytes, &i);
            const b2: u32 = nextByte(bytes, &i);
            const b3: u32 = nextByte(bytes, &i);
            cp = ((b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
        } else {
            cp = 0xFFFD;
        }
        try appendCodePoint(&out, cp);
    }
    return out.toOwnedSlice();
}

/// Render text as a hex string.
///
/// `delimiter` controls how per-byte hex pairs are joined:
///   - .none        -> "48656c6c6f"
///   - .space       -> "48 65 6c 6c 6f"
///   - .prefix_0x   -> "0x48 0x65 ..."
///   - .backslash_x -> "\x48\x65..." (no separators, C-style)
pub fn textToHex(
    allocator: std.mem.Allocator,
    text: []const u8,
    delimiter: Delimiter,
    uppercase: bool,
) ![]u8 {
    const bytes = try utf8Encode(allocator, text);
    defer allocator.free(bytes);

    const digits: []const u8 = if (uppercase) "0123456789ABCDEF" else "0123456789abcdef";

    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();

    for (bytes, 0..) |b, k| {
        if (k > 0 and (delimiter == .space or delimiter == .prefix_0x)) {
            try out.append(' ');
        }
        switch (delimiter) {
            .prefix_0x => try out.appendSlice("0x"),
            .backslash_x => try out.appendSlice("\\x"),
            else => {},
        }
        try out.append(digits[b >> 4]);
        try out.append(digits[b & 0xF]);
    }
    return out.toOwnedSlice();
}

/// Strip common affixes users paste alongside hex — `0x` and `\x` markers
/// (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
/// "aa:bb:cc") — then lowercase. `std.ascii.isWhitespace` covers ASCII
/// whitespace (Python's `\s` also strips exotic Unicode spaces, which cannot
/// contribute valid hex anyway); ASCII lowercasing is sufficient because only
/// [0-9a-f] are valid afterward.
pub fn sanitizeHex(allocator: std.mem.Allocator, input: []const u8) ![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();

    var i: usize = 0;
    while (i < input.len) {
        const c = input[i];
        // Case-insensitive "0x" / "\x" markers consume two bytes.
        if (i + 1 < input.len and (c == '0' or c == '\\') and
            (input[i + 1] == 'x' or input[i + 1] == 'X'))
        {
            i += 2;
            continue;
        }
        if (std.ascii.isWhitespace(c) or c == ',' or c == ':') {
            i += 1;
            continue;
        }
        try out.append(std.ascii.toLower(c));
        i += 1;
    }
    return out.toOwnedSlice();
}

/// Map a single validated hex digit to its numeric value. The fallback arm is
/// unreachable because callers pre-validate the input.
fn hexDigit(c: u8) u8 {
    return switch (c) {
        '0'...'9' => c - '0',
        'a'...'f' => c - 'a' + 10,
        else => 0,
    };
}

/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `err`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
pub fn hexToText(allocator: std.mem.Allocator, hex: []const u8) !DecodeResult {
    const cleaned = try sanitizeHex(allocator, hex);
    defer allocator.free(cleaned);

    if (cleaned.len == 0) return DecodeResult.okText("");

    // After sanitizing + lowercasing, every char must be in [0-9a-f].
    for (cleaned) |c| {
        if (!std.ascii.isHex(c)) {
            return DecodeResult.fail("Hex strings may only contain 0-9 and a-f.");
        }
    }
    if (cleaned.len % 2 != 0) {
        return DecodeResult.fail("Hex must have an even number of digits.");
    }

    const bytes = try allocator.alloc(u8, cleaned.len / 2);
    defer allocator.free(bytes);
    var k: usize = 0;
    while (k < bytes.len) : (k += 1) {
        bytes[k] = hexDigit(cleaned[2 * k]) * 16 + hexDigit(cleaned[2 * k + 1]);
    }
    return DecodeResult.okText(try utf8Decode(allocator, bytes));
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →