Skip to content

Hex ↔ Text Converter — Go source

Convert text to hexadecimal and hex back to text, with delimiter options (none, spaces, 0x, backslash-x) and full UTF-8 support. 100% client-side.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package hexconverter is the Go twin of CosmoDev's src/lib/hexText.ts (dual
// source: the web lib is TypeScript, the CLI lib is Go — kept in lock-step).
// Pure + deterministic, never panics. The table-driven tests in
// hex-converter_test.go share vectors with src/lib/hexText.test.ts so the two
// implementations are held to the same contract.
//
// The conversion mirrors the TS lib exactly: UTF-8 is encoded/decoded by hand
// (so the algorithm is byte-for-byte identical regardless of language runtime),
// and decode of invalid/truncated sequences reproduces the TS behavior —
// missing continuation bytes are read as 0, and an unrecognized lead byte yields
// U+FFFD. The TS hexToText carries an ignored _delimiter parameter for surface
// symmetry with textToHex; this twin drops it, since it has no effect on output.
package hexconverter

import (
	"fmt"
	"regexp"
	"strconv"
	"strings"
)

// Delimiter is the separator style used between encoded bytes. The zero value
// DelimiterNone matches the TS default ('none').
type Delimiter int

const (
	// DelimiterNone concatenates bytes with no separator. It is the zero value,
	// matching the TS default 'none'.
	DelimiterNone Delimiter = iota
	DelimiterSpace
	Delimiter0x
	DelimiterBackslashX
)

// DecodeResult is the Go mirror of the TS DecodeResult. Error is the empty
// string when there is no error (the TS uses null); Ok reports success.
type DecodeResult struct {
	Ok    bool
	Text  string
	Error string
}

var (
	// Sanitization patterns, applied in the same order as sanitizeHex in the TS lib.
	sanitize0x        = regexp.MustCompile(`(?i)0x`)
	sanitizeBackslash = regexp.MustCompile(`(?i)\\x`)
	sanitizeJunk      = regexp.MustCompile(`[\s,:]`)
	// Validation: after sanitizing + lowercasing, only hex digits may remain.
	hexOnly = regexp.MustCompile(`^[0-9a-f]+$`)
)

// Utf8Encode UTF-8 encodes a string into a list of byte values (0..255). It is
// the Go twin of utf8Encode in src/lib/hexText.ts and iterates by code point,
// matching the TS for...of loop.
func Utf8Encode(str string) []byte {
	out := make([]byte, 0, len(str))
	for _, ch := range str {
		cp := uint32(ch)
		switch {
		case cp <= 0x7f:
			out = append(out, byte(cp))
		case cp <= 0x7ff:
			out = append(out, byte(0xc0|(cp>>6)), byte(0x80|(cp&0x3f)))
		case cp <= 0xffff:
			out = append(out,
				byte(0xe0|(cp>>12)),
				byte(0x80|((cp>>6)&0x3f)),
				byte(0x80|(cp&0x3f)),
			)
		default:
			out = append(out,
				byte(0xf0|(cp>>18)),
				byte(0x80|((cp>>12)&0x3f)),
				byte(0x80|((cp>>6)&0x3f)),
				byte(0x80|(cp&0x3f)),
			)
		}
	}
	return out
}

// Utf8Decode UTF-8 decodes bytes to a string; invalid sequences yield U+FFFD.
// It is the Go twin of utf8Decode in src/lib/hexText.ts. Missing continuation
// bytes are read as 0 (mirroring the TS `bytes[i++] ?? 0` nullish coalescing),
// so truncated multibyte leads decode to the same code points as the TS lib.
func Utf8Decode(data []byte) string {
	var b strings.Builder
	i := 0
	for i < len(data) {
		lead := data[i]
		i++
		var cp uint32
		switch {
		case lead <= 0x7f:
			cp = uint32(lead)
		case lead>>5 == 0b110:
			b1 := byte(0)
			if i < len(data) {
				b1 = data[i]
				i++
			}
			cp = (uint32(lead&0x1f) << 6) | uint32(b1&0x3f)
		case lead>>4 == 0b1110:
			b1, b2 := byte(0), byte(0)
			if i < len(data) {
				b1 = data[i]
				i++
			}
			if i < len(data) {
				b2 = data[i]
				i++
			}
			cp = (uint32(lead&0x0f) << 12) | (uint32(b1&0x3f) << 6) | uint32(b2&0x3f)
		case lead>>3 == 0b11110:
			b1, b2, b3 := byte(0), byte(0), byte(0)
			if i < len(data) {
				b1 = data[i]
				i++
			}
			if i < len(data) {
				b2 = data[i]
				i++
			}
			if i < len(data) {
				b3 = data[i]
				i++
			}
			cp = (uint32(lead&0x07) << 18) | (uint32(b1&0x3f) << 12) | (uint32(b2&0x3f) << 6) | uint32(b3&0x3f)
		default:
			cp = 0xfffd
		}
		b.WriteRune(rune(cp))
	}
	return b.String()
}

// TextToHex converts text to hex with a delimiter. It is the Go twin of
// textToHex in src/lib/hexText.ts; the zero values (DelimiterNone, false) match
// the TS defaults ('none', false). uppercase emits A-F instead of a-f.
func TextToHex(text string, delimiter Delimiter, uppercase bool) string {
	enc := Utf8Encode(text)
	hexes := make([]string, len(enc))
	for i, by := range enc {
		hexes[i] = fmt.Sprintf("%02x", by)
	}
	if uppercase {
		for i, h := range hexes {
			hexes[i] = strings.ToUpper(h)
		}
	}
	switch delimiter {
	case DelimiterSpace:
		return strings.Join(hexes, " ")
	case Delimiter0x:
		prefixed := make([]string, len(hexes))
		for i, h := range hexes {
			prefixed[i] = "0x" + h
		}
		return strings.Join(prefixed, " ")
	case DelimiterBackslashX:
		prefixed := make([]string, len(hexes))
		for i, h := range hexes {
			prefixed[i] = "\\x" + h
		}
		return strings.Join(prefixed, "")
	default: // DelimiterNone
		return strings.Join(hexes, "")
	}
}

// SanitizeHex strips 0x, \x, spaces, commas, and colons, then lowercases. It is
// the Go twin of sanitizeHex in src/lib/hexText.ts, applied in the same order.
func SanitizeHex(input string) string {
	s := input
	s = sanitize0x.ReplaceAllString(s, "")
	s = sanitizeBackslash.ReplaceAllString(s, "")
	s = sanitizeJunk.ReplaceAllString(s, "")
	return strings.ToLower(s)
}

// HexToText converts hex to text. It is the Go twin of hexToText in
// src/lib/hexText.ts and reproduces its error messages exactly.
func HexToText(hex string) DecodeResult {
	cleaned := SanitizeHex(hex)
	if len(cleaned) == 0 {
		return DecodeResult{Ok: true, Text: "", Error: ""}
	}
	if !hexOnly.MatchString(cleaned) {
		return DecodeResult{Ok: false, Text: "", Error: "Hex strings may only contain 0-9 and a-f."}
	}
	if len(cleaned)%2 != 0 {
		return DecodeResult{Ok: false, Text: "", Error: "Hex must have an even number of digits."}
	}
	out := make([]byte, 0, len(cleaned)/2)
	for i := 0; i < len(cleaned); i += 2 {
		v, err := strconv.ParseUint(cleaned[i:i+2], 16, 8)
		if err != nil {
			return DecodeResult{Ok: false, Text: "", Error: "Hex strings may only contain 0-9 and a-f."}
		}
		out = append(out, byte(v))
	}
	return DecodeResult{Ok: true, Text: Utf8Decode(out), Error: ""}
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →