Skip to content

Punycode Converter — Go source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package punycode is the Go twin of CosmoDev's src/lib/punycode.ts (dual source:
// the web lib is TypeScript, the CLI lib is Go — kept in lock-step). Pure +
// deterministic, never panics. The table-driven tests in punycode_test.go share
// vectors with src/lib/punycode.test.ts so the two implementations are held to
// the same contract.
//
// Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
// helpers. encodeLabel/decodeLabel operate on a single label (no ACE prefix);
// Encode/Decode wrap them with the "xn--" prefixing and "." splitting of a full
// domain. Mirrors the TS lib's algorithm exactly: code points are iterated as
// runes (Go's native UTF-8 decoding ≡ the TS `for…of` code-point iteration);
// bias adaptation, generalized base-36 digits, and the MAX_SAFE_INTEGER
// overflow guard all map 1:1.
package punycode

import "strings"

// RFC 3492 parameters (section 5), matching the TS constants.
const (
	base        = 36
	tmin        = 1
	tmax        = 26
	skew        = 38
	damp        = 700
	initialBias = 72
	initialN    = 128
	acePrefix   = "xn--"
	// maxInt mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for
	// malformed decode input. The guard makes pathological generalized numbers
	// (e.g. 40 nines) return false instead of running away.
	maxInt = int64(1)<<53 - 1
)

// adapt is the bias adaptation function (RFC 3492 section 6.1). delta and
// numpoints are always non-negative, so Go's integer division matches the TS
// Math.floor divisions exactly.
func adapt(delta, numpoints int64, firsttime bool) int64 {
	var d int64
	if firsttime {
		d = delta / damp
	} else {
		d = delta / 2
	}
	d += d / numpoints
	k := int64(0)
	for d > ((base-tmin)*tmax)/2 {
		d /= (base - tmin)
		k += base
	}
	return k + (base-tmin+1)*d/(d+skew)
}

// digitToChar maps a digit value (0–35) to its RFC 3492 base-36 character
// (lowercase a–z for 0–25, 0–9 for 26–35). d is always in [0,35] on the encode
// path, so the result is a single ASCII byte.
func digitToChar(d int64) byte {
	if d < 26 {
		return byte('a' + d) // a–z
	}
	return byte('0' + (d - 26)) // 0–9
}

// charToDigit maps a byte to its digit value (0–35), case-insensitive, or -1 if
// it is not a valid base-36 digit. Any non-ASCII byte (>= 128) matches no case
// and yields -1, mirroring the TS behavior of rejecting non-ASCII in the
// extension portion.
func charToDigit(c byte) int64 {
	switch {
	case c >= 'a' && c <= 'z':
		return int64(c - 'a')
	case c >= 'A' && c <= 'Z':
		return int64(c - 'A')
	case c >= '0' && c <= '9':
		return int64(c - '0' + 26)
	}
	return -1
}

// hasNonAscii reports whether s contains any non-ASCII code point (>= 128).
// Ranging over runes is equivalent to the TS charCodeAt unit scan: any
// non-ASCII character has a rune/byte value >= 128.
func hasNonAscii(s string) bool {
	for _, r := range s {
		if r >= 128 {
			return true
		}
	}
	return false
}

// EncodeLabel Punycode-encodes a single label (RFC 3492) and returns the
// encoded label with no ACE prefix. Basic (ASCII) code points are emitted
// first, followed by a '-' delimiter (only if there was at least one), then the
// generalized base-36 deltas for the non-basic code points. It is the Go twin
// of encodeLabel() in src/lib/punycode.ts.
func EncodeLabel(input string) string {
	// Iterate by code point (rune) so astral characters (emoji, CJK extensions)
	// are handled as single elements — same as the TS code-point iteration.
	var codePoints []rune
	for _, cp := range input {
		codePoints = append(codePoints, cp)
	}
	length := int64(len(codePoints))

	var output []byte
	for _, cp := range codePoints {
		if cp < 128 {
			output = append(output, byte(cp))
		}
	}
	b := int64(len(output))
	if b > 0 {
		output = append(output, '-')
	}

	n := int64(initialN)
	delta := int64(0)
	bias := int64(initialBias)
	h := b

	for h < length {
		// Smallest code point in the input that is >= n.
		var m int64
		found := false
		for _, cp := range codePoints {
			c := int64(cp)
			if c >= n && (!found || c < m) {
				m = c
				found = true
			}
		}
		delta += (m - n) * (h + 1)
		n = m
		for _, cp := range codePoints {
			c := int64(cp)
			if c < n {
				delta++
			} else if c == n {
				q := delta
				for k := int64(base); ; k += base {
					t := k - bias
					if t < tmin {
						t = tmin
					} else if t > tmax {
						t = tmax
					} // t = max(tmin, min(tmax, k - bias))
					if q < t {
						break
					}
					output = append(output, digitToChar(t+(q-t)%(base-t)))
					q = (q - t) / (base - t)
				}
				output = append(output, digitToChar(q))
				bias = adapt(delta, h+1, h == b)
				delta = 0
				h++
			}
		}
		delta++
		n++
	}

	return string(output)
}

// DecodeLabel Punycode-decodes a single label (RFC 3492). It returns the
// decoded label and true, or ("", false) if the input is malformed (invalid
// digit, truncated generalized number, non-ASCII in the basic portion, or
// arithmetic overflow). It is the Go twin of decodeLabel() in src/lib/punycode.ts,
// where the TS lib returns `string | null`.
func DecodeLabel(input string) (string, bool) {
	lastDash := strings.LastIndex(input, "-")
	var output []rune
	if lastDash >= 0 {
		for i := 0; i < lastDash; i++ {
			if input[i] >= 128 {
				return "", false // basic portion must be ASCII
			}
			output = append(output, rune(input[i]))
		}
	}
	var ext string
	if lastDash >= 0 {
		ext = input[lastDash+1:]
	} else {
		ext = input
	}

	n := int64(initialN)
	i := int64(0)
	bias := int64(initialBias)
	pos := 0

	for pos < len(ext) {
		oldi := i
		w := int64(1)
		for k := int64(base); ; k += base {
			if pos >= len(ext) {
				return "", false // truncated generalized number
			}
			digit := charToDigit(ext[pos])
			if digit < 0 {
				return "", false // invalid digit
			}
			pos++
			if digit >= maxInt/w {
				return "", false // overflow guard
			}
			i += digit * w
			t := k - bias
			if t < tmin {
				t = tmin
			} else if t > tmax {
				t = tmax
			} // t = max(tmin, min(tmax, k - bias))
			if digit < t {
				break
			}
			w *= base - t
		}
		bias = adapt(i-oldi, int64(len(output))+1, oldi == 0)
		outLen := int64(len(output)) + 1
		n += i / outLen
		i %= outLen
		// Insert the decoded code point n at index i (≡ output.splice(i, 0, …)).
		output = append(output, 0)
		copy(output[i+1:], output[i:])
		output[i] = rune(n)
		i++
	}

	return string(output), true
}

// Encode is the IDNA toASCII operation: it encodes a domain to Punycode
// ("xn--") form. It lowercases the whole domain, splits on ".", ACE-encodes
// ("xn--" + Punycode) any label containing a non-ASCII code point, leaves
// ASCII-only labels untouched, and rejoins with ".". Empty input returns empty.
// It is the Go twin of encode() in src/lib/punycode.ts.
func Encode(domain string) string {
	if len(domain) == 0 {
		return ""
	}
	lower := strings.ToLower(domain)
	labels := strings.Split(lower, ".")
	for idx, label := range labels {
		if hasNonAscii(label) {
			labels[idx] = acePrefix + EncodeLabel(label)
		}
	}
	return strings.Join(labels, ".")
}

// Decode is the IDNA toUnicode operation: it decodes a Punycode ("xn--") domain
// back to Unicode. It splits on ".", decodes any label beginning with "xn--"
// (case-insensitive, detected on the lowercased label), leaves every other
// label untouched, and rejoins with ".". It returns the decoded domain and
// true, or ("", false) if any "xn--" label is invalid — the whole domain is
// rejected, matching IDNA semantics. Empty input returns ("", true). It is the
// Go twin of decode() in src/lib/punycode.ts.
func Decode(domain string) (string, bool) {
	if len(domain) == 0 {
		return "", true
	}
	labels := strings.Split(domain, ".")
	out := make([]string, 0, len(labels))
	for _, label := range labels {
		if strings.HasPrefix(strings.ToLower(label), acePrefix) && len(label) > len(acePrefix) {
			decoded, ok := DecodeLabel(label[len(acePrefix):])
			if !ok {
				return "", false
			}
			out = append(out, decoded)
		} else {
			out = append(out, label)
		}
	}
	return strings.Join(out, "."), true
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →