Hex ↔ Text Converter — Zig source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! hex-converter — pure hex ↔ text conversion.
//!
//! Language: Zig 0.13 (standard library only)
//! Source: CosmoDev polyglot showcase port of the hex-converter tool,
//! ported from src/lib/hexText.ts (the canonical TypeScript
//! implementation).
//! License: display source — part of CosmoDev's polyglot tool pages
//! (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//! byte sequences decode to U+FFFD, matching the canonical logic.
const std = @import("std");
/// How encoded bytes are joined when rendered as a hex string.
pub const Delimiter = enum {
/// No separator: "48656c6c6f".
none,
/// Single space between bytes: "48 65 6c 6c 6f".
space,
/// Each byte prefixed with "0x", space-separated.
prefix_0x,
/// Each byte prefixed with "\x", no separator (C-style).
backslash_x,
};
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// `ok`, `text`, and `err` (null when ok; `error` is a Zig keyword).
///
/// On success `text` is allocated with the caller's allocator and owned by
/// the caller (empty input yields a static empty literal — freeing a
/// zero-length slice is a no-op in Zig, so `allocator.free(result.text)` is
/// always safe).
pub const DecodeResult = struct {
ok: bool,
text: []const u8,
err: ?[]const u8,
pub fn okText(text: []const u8) DecodeResult {
return .{ .ok = true, .text = text, .err = null };
}
pub fn fail(message: []const u8) DecodeResult {
return .{ .ok = false, .text = "", .err = message };
}
};
/// U+FFFD in UTF-8 (EF BF BD), substituted for malformed sequences.
const replacement_utf8 = [3]u8{ 0xEF, 0xBF, 0xBD };
/// Read the next byte, returning 0 past end-of-input (the canonical decoder's
/// behavior) and advancing the cursor.
fn nextByte(bytes: []const u8, i: *usize) u8 {
if (i.* >= bytes.len) return 0;
const b = bytes[i.*];
i.* += 1;
return b;
}
/// UTF-8 encode a string into bytes.
///
/// Zig string literals are already UTF-8; the code points are re-derived and
/// pushed through the canonical 1..4-byte branches below so every language in
/// the polyglot showcase produces byte-identical output. Input is required to
/// be valid UTF-8 (Zig string literals always are).
pub fn utf8Encode(allocator: std.mem.Allocator, text: []const u8) ![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
const view = std.unicode.Utf8View.initUnchecked(text);
var it = view.iterator();
while (it.nextCodePoint()) |cp| {
if (cp <= 0x7F) {
try out.append(@intCast(cp));
} else if (cp <= 0x7FF) {
try out.append(@intCast(0xC0 | (cp >> 6)));
try out.append(@intCast(0x80 | (cp & 0x3F)));
} else if (cp <= 0xFFFF) {
try out.append(@intCast(0xE0 | (cp >> 12)));
try out.append(@intCast(0x80 | ((cp >> 6) & 0x3F)));
try out.append(@intCast(0x80 | (cp & 0x3F)));
} else {
try out.append(@intCast(0xF0 | (cp >> 18)));
try out.append(@intCast(0x80 | ((cp >> 12) & 0x3F)));
try out.append(@intCast(0x80 | ((cp >> 6) & 0x3F)));
try out.append(@intCast(0x80 | (cp & 0x3F)));
}
}
return out.toOwnedSlice();
}
/// Append one code point as UTF-8, substituting U+FFFD for surrogates and
/// out-of-range values — `utf8Encode`'s error path made total, so the decoder
/// never rejects.
fn appendCodePoint(out: *std.ArrayList(u8), cp: u32) !void {
if (cp > 0x10FFFF or (cp >= 0xD800 and cp <= 0xDFFF)) {
try out.appendSlice(&replacement_utf8);
return;
}
var buf: [4]u8 = undefined;
const n = try std.unicode.utf8Encode(@intCast(cp), &buf);
try out.appendSlice(buf[0..n]);
}
/// UTF-8 decode a byte slice into a string (UTF-8). Truncated or invalid
/// sequences yield U+FFFD; missing continuation bytes are taken as 0,
/// matching the canonical decoder's lenient consumption.
pub fn utf8Decode(allocator: std.mem.Allocator, bytes: []const u8) ![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var i: usize = 0;
while (i < bytes.len) {
const b: u32 = bytes[i];
i += 1;
var cp: u32 = undefined;
if (b <= 0x7F) {
cp = b;
} else if (b >> 5 == 0b110) {
const b1: u32 = nextByte(bytes, &i);
cp = ((b & 0x1F) << 6) | (b1 & 0x3F);
} else if (b >> 4 == 0b1110) {
const b1: u32 = nextByte(bytes, &i);
const b2: u32 = nextByte(bytes, &i);
cp = ((b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
} else if (b >> 3 == 0b11110) {
const b1: u32 = nextByte(bytes, &i);
const b2: u32 = nextByte(bytes, &i);
const b3: u32 = nextByte(bytes, &i);
cp = ((b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
} else {
cp = 0xFFFD;
}
try appendCodePoint(&out, cp);
}
return out.toOwnedSlice();
}
/// Render text as a hex string.
///
/// `delimiter` controls how per-byte hex pairs are joined:
/// - .none -> "48656c6c6f"
/// - .space -> "48 65 6c 6c 6f"
/// - .prefix_0x -> "0x48 0x65 ..."
/// - .backslash_x -> "\x48\x65..." (no separators, C-style)
pub fn textToHex(
allocator: std.mem.Allocator,
text: []const u8,
delimiter: Delimiter,
uppercase: bool,
) ![]u8 {
const bytes = try utf8Encode(allocator, text);
defer allocator.free(bytes);
const digits: []const u8 = if (uppercase) "0123456789ABCDEF" else "0123456789abcdef";
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
for (bytes, 0..) |b, k| {
if (k > 0 and (delimiter == .space or delimiter == .prefix_0x)) {
try out.append(' ');
}
switch (delimiter) {
.prefix_0x => try out.appendSlice("0x"),
.backslash_x => try out.appendSlice("\\x"),
else => {},
}
try out.append(digits[b >> 4]);
try out.append(digits[b & 0xF]);
}
return out.toOwnedSlice();
}
/// Strip common affixes users paste alongside hex — `0x` and `\x` markers
/// (case-insensitive, anywhere), whitespace, commas, and colons (MAC-style
/// "aa:bb:cc") — then lowercase. `std.ascii.isWhitespace` covers ASCII
/// whitespace (Python's `\s` also strips exotic Unicode spaces, which cannot
/// contribute valid hex anyway); ASCII lowercasing is sufficient because only
/// [0-9a-f] are valid afterward.
pub fn sanitizeHex(allocator: std.mem.Allocator, input: []const u8) ![]u8 {
var out = std.ArrayList(u8).init(allocator);
errdefer out.deinit();
var i: usize = 0;
while (i < input.len) {
const c = input[i];
// Case-insensitive "0x" / "\x" markers consume two bytes.
if (i + 1 < input.len and (c == '0' or c == '\\') and
(input[i + 1] == 'x' or input[i + 1] == 'X'))
{
i += 2;
continue;
}
if (std.ascii.isWhitespace(c) or c == ',' or c == ':') {
i += 1;
continue;
}
try out.append(std.ascii.toLower(c));
i += 1;
}
return out.toOwnedSlice();
}
/// Map a single validated hex digit to its numeric value. The fallback arm is
/// unreachable because callers pre-validate the input.
fn hexDigit(c: u8) u8 {
return switch (c) {
'0'...'9' => c - '0',
'a'...'f' => c - 'a' + 10,
else => 0,
};
}
/// Decode a (possibly decorated) hex string back to text. Invalid characters
/// and odd lengths are reported via `err`; valid input containing malformed
/// UTF-8 still decodes with U+FFFD substitution.
pub fn hexToText(allocator: std.mem.Allocator, hex: []const u8) !DecodeResult {
const cleaned = try sanitizeHex(allocator, hex);
defer allocator.free(cleaned);
if (cleaned.len == 0) return DecodeResult.okText("");
// After sanitizing + lowercasing, every char must be in [0-9a-f].
for (cleaned) |c| {
if (!std.ascii.isHex(c)) {
return DecodeResult.fail("Hex strings may only contain 0-9 and a-f.");
}
}
if (cleaned.len % 2 != 0) {
return DecodeResult.fail("Hex must have an even number of digits.");
}
const bytes = try allocator.alloc(u8, cleaned.len / 2);
defer allocator.free(bytes);
var k: usize = 0;
while (k < bytes.len) : (k += 1) {
bytes[k] = hexDigit(cleaned[2 * k]) * 16 + hexDigit(cleaned[2 * k + 1]);
}
return DecodeResult.okText(try utf8Decode(allocator, bytes));
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →