Hex ↔ Text Converter — C# source
Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// hex-converter — pure hex ↔ text conversion.
//
// Language: C# (.NET 8+ / C# 12, standard library only)
// Source: CosmoDev polyglot showcase port of the hex-converter tool,
// ported from src/lib/hexText.ts (the canonical TypeScript
// implementation).
// License: display source — part of CosmoDev's polyglot tool pages
// (dev.cosmolabs.org). Deterministic, side-effect free; invalid
// byte sequences decode to U+FFFD, matching the canonical logic.
using System;
using System.Collections.Generic;
using System.Linq;
using System.Text;
using System.Text.RegularExpressions;
namespace CosmoDev.HexConverter;
/// <summary>How encoded bytes are joined when rendered as a hex string.</summary>
public enum Delimiter
{
/// <summary>No separator: "48656c6c6f".</summary>
None,
/// <summary>Single space between bytes: "48 65 6c 6c 6f".</summary>
Space,
/// <summary>Each byte prefixed with "0x", space-separated.</summary>
Prefix0x,
/// <summary>Each byte prefixed with "\x", no separator (C-style).</summary>
BackslashX,
}
/// <summary>
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// <c>Ok</c>, <c>Text</c>, and <c>Error</c> (null when Ok).
/// </summary>
/// <param name="Ok">Whether the input decoded successfully.</param>
/// <param name="Text">The decoded text (empty on failure).</param>
/// <param name="Error">Failure reason (null on success).</param>
public sealed record DecodeResult(bool Ok, string Text, string? Error)
{
public static DecodeResult OkText(string text) => new(true, text, null);
public static DecodeResult Fail(string message) => new(false, string.Empty, message);
}
public static class HexConverter
{
/// <summary>U+FFFD, substituted for malformed UTF-8 on decode.</summary>
private const string ReplacementChar = "�";
/// <summary>
/// UTF-8 encode a string into a byte array. Hand-rolled for byte-exact
/// parity across every showcase language; <see cref="string.EnumerateRunes"/>
/// iterates by Unicode scalar value (invalid sequences yield U+FFFD), so
/// astral characters encode as 4-byte sequences.
/// </summary>
public static byte[] Utf8Encode(string text)
{
var bytes = new List<byte>();
foreach (Rune rune in text.EnumerateRunes())
{
uint cp = (uint)rune.Value;
if (cp <= 0x7F)
{
bytes.Add((byte)cp);
}
else if (cp <= 0x7FF)
{
bytes.Add((byte)(0xC0 | (cp >> 6)));
bytes.Add((byte)(0x80 | (cp & 0x3F)));
}
else if (cp <= 0xFFFF)
{
bytes.Add((byte)(0xE0 | (cp >> 12)));
bytes.Add((byte)(0x80 | ((cp >> 6) & 0x3F)));
bytes.Add((byte)(0x80 | (cp & 0x3F)));
}
else
{
bytes.Add((byte)(0xF0 | (cp >> 18)));
bytes.Add((byte)(0x80 | ((cp >> 12) & 0x3F)));
bytes.Add((byte)(0x80 | ((cp >> 6) & 0x3F)));
bytes.Add((byte)(0x80 | (cp & 0x3F)));
}
}
return bytes.ToArray();
}
/// <summary>
/// UTF-8 decode a byte array into a string. Truncated or invalid sequences
/// yield U+FFFD; missing continuation bytes are taken as 0, matching the
/// canonical decoder's lenient consumption. Surrogate / out-of-range code
/// points are also mapped to U+FFFD so the decoder is total.
/// </summary>
public static string Utf8Decode(byte[] bytes)
{
var outText = new StringBuilder();
int i = 0;
// Reads past the end return 0 — the canonical decoder's behavior.
byte NextByte() => i < bytes.Length ? bytes[i++] : (byte)0;
while (i < bytes.Length)
{
byte b = bytes[i++];
uint cp;
if (b <= 0x7F)
{
cp = b;
}
else if ((b >> 5) == 0b110)
{
uint b1 = NextByte();
cp = ((uint)(b & 0x1F) << 6) | (b1 & 0x3F);
}
else if ((b >> 4) == 0b1110)
{
uint b1 = NextByte();
uint b2 = NextByte();
cp = ((uint)(b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
}
else if ((b >> 3) == 0b11110)
{
uint b1 = NextByte();
uint b2 = NextByte();
uint b3 = NextByte();
cp = ((uint)(b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
}
else
{
cp = 0xFFFD;
}
outText.Append(
cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)
? ReplacementChar
: char.ConvertFromUtf32((int)cp));
}
return outText.ToString();
}
/// <summary>
/// Render text as a hex string.
///
/// <paramref name="delimiter"/> controls how per-byte hex pairs are joined:
/// - <see cref="Delimiter.None"/> -> "48656c6c6f"
/// - <see cref="Delimiter.Space"/> -> "48 65 6c 6c 6f"
/// - <see cref="Delimiter.Prefix0x"/> -> "0x48 0x65 ..."
/// - <see cref="Delimiter.BackslashX"/> -> "\x48\x65..." (no separators, C-style)
/// </summary>
public static string TextToHex(string text, Delimiter delimiter = Delimiter.None, bool uppercase = false)
{
List<string> hexes = Utf8Encode(text).Select(b => b.ToString("x2")).ToList();
if (uppercase)
{
hexes = hexes.Select(h => h.ToUpperInvariant()).ToList();
}
return delimiter switch
{
Delimiter.Space => string.Join(" ", hexes),
Delimiter.Prefix0x => string.Join(" ", hexes.Select(h => $"0x{h}")),
Delimiter.BackslashX => string.Concat(hexes.Select(h => $"\\x{h}")),
_ => string.Concat(hexes),
};
}
/// <summary>
/// Strip common affixes users paste alongside hex — <c>0x</c> and <c>\x</c>
/// markers (case-insensitive, anywhere), whitespace, commas, and colons
/// (MAC-style "aa:bb:cc") — then lowercase. .NET's <c>\s</c> is
/// Unicode-aware, matching the canonical TS regex.
/// </summary>
public static string SanitizeHex(string? input)
{
if (string.IsNullOrEmpty(input)) return string.Empty;
string noMarkers = Regex.Replace(input, "0x", "", RegexOptions.IgnoreCase);
noMarkers = Regex.Replace(noMarkers, @"\\x", "", RegexOptions.IgnoreCase);
string cleaned = Regex.Replace(noMarkers, @"[\s,:]", "");
return cleaned.ToLowerInvariant();
}
/// <summary>
/// Decode a (possibly decorated) hex string back to text. Invalid
/// characters and odd lengths are reported via <c>Error</c>; valid input
/// containing malformed UTF-8 still decodes with U+FFFD substitution.
/// </summary>
public static DecodeResult HexToText(string? hex)
{
string cleaned = SanitizeHex(hex);
if (cleaned.Length == 0) return DecodeResult.OkText(string.Empty);
// After sanitizing + lowercasing, every char must be in [0-9a-f].
if (!Regex.IsMatch(cleaned, "^[0-9a-f]+$"))
{
return DecodeResult.Fail("Hex strings may only contain 0-9 and a-f.");
}
if (cleaned.Length % 2 != 0)
{
return DecodeResult.Fail("Hex must have an even number of digits.");
}
var bytes = new byte[cleaned.Length / 2];
for (int i = 0; i < bytes.Length; i++)
{
bytes[i] = Convert.ToByte(cleaned.Substring(2 * i, 2), 16);
}
return DecodeResult.OkText(Utf8Decode(bytes));
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →