Skip to content

Hex ↔ Text Converter — C# source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// hex-converter — pure hex ↔ text conversion.
//
// Language: C# (.NET 8+ / C# 12, standard library only)
// Source:   CosmoDev polyglot showcase port of the hex-converter tool,
//           ported from src/lib/hexText.ts (the canonical TypeScript
//           implementation).
// License:  display source — part of CosmoDev's polyglot tool pages
//           (dev.cosmolabs.org). Deterministic, side-effect free; invalid
//           byte sequences decode to U+FFFD, matching the canonical logic.

using System;
using System.Collections.Generic;
using System.Linq;
using System.Text;
using System.Text.RegularExpressions;

namespace CosmoDev.HexConverter;

/// <summary>How encoded bytes are joined when rendered as a hex string.</summary>
public enum Delimiter
{
    /// <summary>No separator: "48656c6c6f".</summary>
    None,

    /// <summary>Single space between bytes: "48 65 6c 6c 6f".</summary>
    Space,

    /// <summary>Each byte prefixed with "0x", space-separated.</summary>
    Prefix0x,

    /// <summary>Each byte prefixed with "\x", no separator (C-style).</summary>
    BackslashX,
}

/// <summary>
/// Outcome of decoding hex back to text. Mirrors the canonical TS surface:
/// <c>Ok</c>, <c>Text</c>, and <c>Error</c> (null when Ok).
/// </summary>
/// <param name="Ok">Whether the input decoded successfully.</param>
/// <param name="Text">The decoded text (empty on failure).</param>
/// <param name="Error">Failure reason (null on success).</param>
public sealed record DecodeResult(bool Ok, string Text, string? Error)
{
    public static DecodeResult OkText(string text) => new(true, text, null);

    public static DecodeResult Fail(string message) => new(false, string.Empty, message);
}

public static class HexConverter
{
    /// <summary>U+FFFD, substituted for malformed UTF-8 on decode.</summary>
    private const string ReplacementChar = "�";

    /// <summary>
    /// UTF-8 encode a string into a byte array. Hand-rolled for byte-exact
    /// parity across every showcase language; <see cref="string.EnumerateRunes"/>
    /// iterates by Unicode scalar value (invalid sequences yield U+FFFD), so
    /// astral characters encode as 4-byte sequences.
    /// </summary>
    public static byte[] Utf8Encode(string text)
    {
        var bytes = new List<byte>();
        foreach (Rune rune in text.EnumerateRunes())
        {
            uint cp = (uint)rune.Value;
            if (cp <= 0x7F)
            {
                bytes.Add((byte)cp);
            }
            else if (cp <= 0x7FF)
            {
                bytes.Add((byte)(0xC0 | (cp >> 6)));
                bytes.Add((byte)(0x80 | (cp & 0x3F)));
            }
            else if (cp <= 0xFFFF)
            {
                bytes.Add((byte)(0xE0 | (cp >> 12)));
                bytes.Add((byte)(0x80 | ((cp >> 6) & 0x3F)));
                bytes.Add((byte)(0x80 | (cp & 0x3F)));
            }
            else
            {
                bytes.Add((byte)(0xF0 | (cp >> 18)));
                bytes.Add((byte)(0x80 | ((cp >> 12) & 0x3F)));
                bytes.Add((byte)(0x80 | ((cp >> 6) & 0x3F)));
                bytes.Add((byte)(0x80 | (cp & 0x3F)));
            }
        }
        return bytes.ToArray();
    }

    /// <summary>
    /// UTF-8 decode a byte array into a string. Truncated or invalid sequences
    /// yield U+FFFD; missing continuation bytes are taken as 0, matching the
    /// canonical decoder's lenient consumption. Surrogate / out-of-range code
    /// points are also mapped to U+FFFD so the decoder is total.
    /// </summary>
    public static string Utf8Decode(byte[] bytes)
    {
        var outText = new StringBuilder();
        int i = 0;

        // Reads past the end return 0 — the canonical decoder's behavior.
        byte NextByte() => i < bytes.Length ? bytes[i++] : (byte)0;

        while (i < bytes.Length)
        {
            byte b = bytes[i++];
            uint cp;
            if (b <= 0x7F)
            {
                cp = b;
            }
            else if ((b >> 5) == 0b110)
            {
                uint b1 = NextByte();
                cp = ((uint)(b & 0x1F) << 6) | (b1 & 0x3F);
            }
            else if ((b >> 4) == 0b1110)
            {
                uint b1 = NextByte();
                uint b2 = NextByte();
                cp = ((uint)(b & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F);
            }
            else if ((b >> 3) == 0b11110)
            {
                uint b1 = NextByte();
                uint b2 = NextByte();
                uint b3 = NextByte();
                cp = ((uint)(b & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
            }
            else
            {
                cp = 0xFFFD;
            }

            outText.Append(
                cp > 0x10FFFF || (cp >= 0xD800 && cp <= 0xDFFF)
                    ? ReplacementChar
                    : char.ConvertFromUtf32((int)cp));
        }
        return outText.ToString();
    }

    /// <summary>
    /// Render text as a hex string.
    ///
    /// <paramref name="delimiter"/> controls how per-byte hex pairs are joined:
    ///   - <see cref="Delimiter.None"/>       -> "48656c6c6f"
    ///   - <see cref="Delimiter.Space"/>      -> "48 65 6c 6c 6f"
    ///   - <see cref="Delimiter.Prefix0x"/>   -> "0x48 0x65 ..."
    ///   - <see cref="Delimiter.BackslashX"/> -> "\x48\x65..." (no separators, C-style)
    /// </summary>
    public static string TextToHex(string text, Delimiter delimiter = Delimiter.None, bool uppercase = false)
    {
        List<string> hexes = Utf8Encode(text).Select(b => b.ToString("x2")).ToList();
        if (uppercase)
        {
            hexes = hexes.Select(h => h.ToUpperInvariant()).ToList();
        }
        return delimiter switch
        {
            Delimiter.Space => string.Join(" ", hexes),
            Delimiter.Prefix0x => string.Join(" ", hexes.Select(h => $"0x{h}")),
            Delimiter.BackslashX => string.Concat(hexes.Select(h => $"\\x{h}")),
            _ => string.Concat(hexes),
        };
    }

    /// <summary>
    /// Strip common affixes users paste alongside hex — <c>0x</c> and <c>\x</c>
    /// markers (case-insensitive, anywhere), whitespace, commas, and colons
    /// (MAC-style "aa:bb:cc") — then lowercase. .NET's <c>\s</c> is
    /// Unicode-aware, matching the canonical TS regex.
    /// </summary>
    public static string SanitizeHex(string? input)
    {
        if (string.IsNullOrEmpty(input)) return string.Empty;

        string noMarkers = Regex.Replace(input, "0x", "", RegexOptions.IgnoreCase);
        noMarkers = Regex.Replace(noMarkers, @"\\x", "", RegexOptions.IgnoreCase);
        string cleaned = Regex.Replace(noMarkers, @"[\s,:]", "");
        return cleaned.ToLowerInvariant();
    }

    /// <summary>
    /// Decode a (possibly decorated) hex string back to text. Invalid
    /// characters and odd lengths are reported via <c>Error</c>; valid input
    /// containing malformed UTF-8 still decodes with U+FFFD substitution.
    /// </summary>
    public static DecodeResult HexToText(string? hex)
    {
        string cleaned = SanitizeHex(hex);
        if (cleaned.Length == 0) return DecodeResult.OkText(string.Empty);

        // After sanitizing + lowercasing, every char must be in [0-9a-f].
        if (!Regex.IsMatch(cleaned, "^[0-9a-f]+$"))
        {
            return DecodeResult.Fail("Hex strings may only contain 0-9 and a-f.");
        }
        if (cleaned.Length % 2 != 0)
        {
            return DecodeResult.Fail("Hex must have an even number of digits.");
        }

        var bytes = new byte[cleaned.Length / 2];
        for (int i = 0; i < bytes.Length; i++)
        {
            bytes[i] = Convert.ToByte(cleaned.Substring(2 * i, 2), 16);
        }
        return DecodeResult.OkText(Utf8Decode(bytes));
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →