Punycode Converter — Go source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package punycode is the Go twin of CosmoDev's src/lib/punycode.ts (dual source:
// the web lib is TypeScript, the CLI lib is Go — kept in lock-step). Pure +
// deterministic, never panics. The table-driven tests in punycode_test.go share
// vectors with src/lib/punycode.test.ts so the two implementations are held to
// the same contract.
//
// Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
// helpers. encodeLabel/decodeLabel operate on a single label (no ACE prefix);
// Encode/Decode wrap them with the "xn--" prefixing and "." splitting of a full
// domain. Mirrors the TS lib's algorithm exactly: code points are iterated as
// runes (Go's native UTF-8 decoding ≡ the TS `for…of` code-point iteration);
// bias adaptation, generalized base-36 digits, and the MAX_SAFE_INTEGER
// overflow guard all map 1:1.
package punycode
import "strings"
// RFC 3492 parameters (section 5), matching the TS constants.
const (
base = 36
tmin = 1
tmax = 26
skew = 38
damp = 700
initialBias = 72
initialN = 128
acePrefix = "xn--"
// maxInt mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for
// malformed decode input. The guard makes pathological generalized numbers
// (e.g. 40 nines) return false instead of running away.
maxInt = int64(1)<<53 - 1
)
// adapt is the bias adaptation function (RFC 3492 section 6.1). delta and
// numpoints are always non-negative, so Go's integer division matches the TS
// Math.floor divisions exactly.
func adapt(delta, numpoints int64, firsttime bool) int64 {
var d int64
if firsttime {
d = delta / damp
} else {
d = delta / 2
}
d += d / numpoints
k := int64(0)
for d > ((base-tmin)*tmax)/2 {
d /= (base - tmin)
k += base
}
return k + (base-tmin+1)*d/(d+skew)
}
// digitToChar maps a digit value (0–35) to its RFC 3492 base-36 character
// (lowercase a–z for 0–25, 0–9 for 26–35). d is always in [0,35] on the encode
// path, so the result is a single ASCII byte.
func digitToChar(d int64) byte {
if d < 26 {
return byte('a' + d) // a–z
}
return byte('0' + (d - 26)) // 0–9
}
// charToDigit maps a byte to its digit value (0–35), case-insensitive, or -1 if
// it is not a valid base-36 digit. Any non-ASCII byte (>= 128) matches no case
// and yields -1, mirroring the TS behavior of rejecting non-ASCII in the
// extension portion.
func charToDigit(c byte) int64 {
switch {
case c >= 'a' && c <= 'z':
return int64(c - 'a')
case c >= 'A' && c <= 'Z':
return int64(c - 'A')
case c >= '0' && c <= '9':
return int64(c - '0' + 26)
}
return -1
}
// hasNonAscii reports whether s contains any non-ASCII code point (>= 128).
// Ranging over runes is equivalent to the TS charCodeAt unit scan: any
// non-ASCII character has a rune/byte value >= 128.
func hasNonAscii(s string) bool {
for _, r := range s {
if r >= 128 {
return true
}
}
return false
}
// EncodeLabel Punycode-encodes a single label (RFC 3492) and returns the
// encoded label with no ACE prefix. Basic (ASCII) code points are emitted
// first, followed by a '-' delimiter (only if there was at least one), then the
// generalized base-36 deltas for the non-basic code points. It is the Go twin
// of encodeLabel() in src/lib/punycode.ts.
func EncodeLabel(input string) string {
// Iterate by code point (rune) so astral characters (emoji, CJK extensions)
// are handled as single elements — same as the TS code-point iteration.
var codePoints []rune
for _, cp := range input {
codePoints = append(codePoints, cp)
}
length := int64(len(codePoints))
var output []byte
for _, cp := range codePoints {
if cp < 128 {
output = append(output, byte(cp))
}
}
b := int64(len(output))
if b > 0 {
output = append(output, '-')
}
n := int64(initialN)
delta := int64(0)
bias := int64(initialBias)
h := b
for h < length {
// Smallest code point in the input that is >= n.
var m int64
found := false
for _, cp := range codePoints {
c := int64(cp)
if c >= n && (!found || c < m) {
m = c
found = true
}
}
delta += (m - n) * (h + 1)
n = m
for _, cp := range codePoints {
c := int64(cp)
if c < n {
delta++
} else if c == n {
q := delta
for k := int64(base); ; k += base {
t := k - bias
if t < tmin {
t = tmin
} else if t > tmax {
t = tmax
} // t = max(tmin, min(tmax, k - bias))
if q < t {
break
}
output = append(output, digitToChar(t+(q-t)%(base-t)))
q = (q - t) / (base - t)
}
output = append(output, digitToChar(q))
bias = adapt(delta, h+1, h == b)
delta = 0
h++
}
}
delta++
n++
}
return string(output)
}
// DecodeLabel Punycode-decodes a single label (RFC 3492). It returns the
// decoded label and true, or ("", false) if the input is malformed (invalid
// digit, truncated generalized number, non-ASCII in the basic portion, or
// arithmetic overflow). It is the Go twin of decodeLabel() in src/lib/punycode.ts,
// where the TS lib returns `string | null`.
func DecodeLabel(input string) (string, bool) {
lastDash := strings.LastIndex(input, "-")
var output []rune
if lastDash >= 0 {
for i := 0; i < lastDash; i++ {
if input[i] >= 128 {
return "", false // basic portion must be ASCII
}
output = append(output, rune(input[i]))
}
}
var ext string
if lastDash >= 0 {
ext = input[lastDash+1:]
} else {
ext = input
}
n := int64(initialN)
i := int64(0)
bias := int64(initialBias)
pos := 0
for pos < len(ext) {
oldi := i
w := int64(1)
for k := int64(base); ; k += base {
if pos >= len(ext) {
return "", false // truncated generalized number
}
digit := charToDigit(ext[pos])
if digit < 0 {
return "", false // invalid digit
}
pos++
if digit >= maxInt/w {
return "", false // overflow guard
}
i += digit * w
t := k - bias
if t < tmin {
t = tmin
} else if t > tmax {
t = tmax
} // t = max(tmin, min(tmax, k - bias))
if digit < t {
break
}
w *= base - t
}
bias = adapt(i-oldi, int64(len(output))+1, oldi == 0)
outLen := int64(len(output)) + 1
n += i / outLen
i %= outLen
// Insert the decoded code point n at index i (≡ output.splice(i, 0, …)).
output = append(output, 0)
copy(output[i+1:], output[i:])
output[i] = rune(n)
i++
}
return string(output), true
}
// Encode is the IDNA toASCII operation: it encodes a domain to Punycode
// ("xn--") form. It lowercases the whole domain, splits on ".", ACE-encodes
// ("xn--" + Punycode) any label containing a non-ASCII code point, leaves
// ASCII-only labels untouched, and rejoins with ".". Empty input returns empty.
// It is the Go twin of encode() in src/lib/punycode.ts.
func Encode(domain string) string {
if len(domain) == 0 {
return ""
}
lower := strings.ToLower(domain)
labels := strings.Split(lower, ".")
for idx, label := range labels {
if hasNonAscii(label) {
labels[idx] = acePrefix + EncodeLabel(label)
}
}
return strings.Join(labels, ".")
}
// Decode is the IDNA toUnicode operation: it decodes a Punycode ("xn--") domain
// back to Unicode. It splits on ".", decodes any label beginning with "xn--"
// (case-insensitive, detected on the lowercased label), leaves every other
// label untouched, and rejoins with ".". It returns the decoded domain and
// true, or ("", false) if any "xn--" label is invalid — the whole domain is
// rejected, matching IDNA semantics. Empty input returns ("", true). It is the
// Go twin of decode() in src/lib/punycode.ts.
func Decode(domain string) (string, bool) {
if len(domain) == 0 {
return "", true
}
labels := strings.Split(domain, ".")
out := make([]string, 0, len(labels))
for _, label := range labels {
if strings.HasPrefix(strings.ToLower(label), acePrefix) && len(label) > len(acePrefix) {
decoded, ok := DecodeLabel(label[len(acePrefix):])
if !ok {
return "", false
}
out = append(out, decoded)
} else {
out = append(out, label)
}
}
return strings.Join(out, "."), true
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →