Punycode Converter — Python source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""punycode — RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
Language: Python (3.9+, standard library only)
Source: CosmoDev polyglot showcase port of the Punycode tool, ported from
src/lib/punycode.ts (canonical TypeScript) and cli/punycode/punycode.go
(the live Go CLI twin — the two are kept in lock-step).
License: display source — part of CosmoDev's polyglot tool pages.
Design goals:
- Pure + deterministic; never raises (decode returns Optional[str], None on
malformed input).
- Functionally equivalent to the Go/TS reference: same inputs -> same outputs.
- Self-contained: stdlib only (no pip packages).
Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
helpers. encode_label/decode_label operate on a single label (no ACE prefix);
encode/decode wrap them with the "xn--" prefixing and "." splitting of a full
domain. Python strings are Unicode code-point sequences (≡ Go runes ≡ the TS
code-point iteration), so bias adaptation, generalized base-36 digits, and the
2^53-1 overflow guard map 1:1. Python's // is floor division; every operand
here is non-negative, so it matches Go's truncated integer division exactly.
"""
from __future__ import annotations
from typing import List, Optional
__all__ = ["encode", "decode", "encode_label", "decode_label"]
# RFC 3492 parameters (section 5), matching the TS/Go constants.
BASE = 36
TMIN = 1
TMAX = 26
SKEW = 38
DAMP = 700
INITIAL_BIAS = 72
INITIAL_N = 128
ACE_PREFIX = "xn--"
# Mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for malformed
# decode input. Makes pathological generalized numbers (e.g. 40 nines) return
# None instead of running away.
MAX_INT = (1 << 53) - 1
def _adapt(delta: int, numpoints: int, firsttime: bool) -> int:
"""Bias adaptation (RFC 3492 section 6.1). delta/numpoints are always
non-negative, so // matches the reference integer division exactly."""
d = (delta // DAMP) if firsttime else (delta // 2)
d += d // numpoints
k = 0
while d > ((BASE - TMIN) * TMAX) // 2:
d //= BASE - TMIN
k += BASE
return k + (BASE - TMIN + 1) * d // (d + SKEW)
def _digit_to_char(d: int) -> str:
"""Map a digit value (0–35) to its RFC 3492 base-36 character (lowercase
a–z for 0–25, 0–9 for 26–35). d is always in [0,35] on the encode path."""
return chr(ord("a") + d) if d < 26 else chr(ord("0") + (d - 26))
def _char_to_digit(c: str) -> int:
"""Map a character to its digit value (0–35), case-insensitive, or -1 if it
is not a valid base-36 digit. Any non-ASCII char yields -1, mirroring the
reference behavior of rejecting non-ASCII in the extension portion."""
code = ord(c)
if 0x61 <= code <= 0x7A: # a–z
return code - 0x61
if 0x41 <= code <= 0x5A: # A–Z
return code - 0x41
if 0x30 <= code <= 0x39: # 0–9
return code - 0x30 + 26
return -1
def _has_non_ascii(s: str) -> bool:
"""True if the string contains any non-ASCII code point (>= 128)."""
return any(ord(c) >= 128 for c in s)
def _cp_to_char(n: int) -> str:
"""Safely turn a decoded code point into a character. Mirrors Go's rune(n)
(which emits U+FFFD for an out-of-range or surrogate value) rather than
raising."""
return chr(n) if 0 <= n <= 0x10FFFF else "�"
def encode_label(input: str) -> str:
"""Punycode-encode a single label (RFC 3492) and return the encoded label
with no ACE prefix. Basic (ASCII) code points are emitted first, followed by
a '-' delimiter (only if there was at least one), then the generalized
base-36 deltas for the non-basic code points. Twin of encodeLabel() in the
TS/Go."""
code_points = list(input) # iterate by code point (astral chars are single)
length = len(code_points)
output: List[str] = [c for c in code_points if ord(c) < 128]
b = len(output)
if b > 0:
output.append("-")
n = INITIAL_N
delta = 0
bias = INITIAL_BIAS
h = b
while h < length:
# Smallest code point in the input that is >= n. The while guard ensures
# at least one such code point remains, so min() never gets an empty
# iterable. Single-char comparison is by code point (≡ by ord).
m = min(c for c in code_points if ord(c) >= n)
delta += (ord(m) - n) * (h + 1)
n = ord(m)
for c in code_points:
code = ord(c)
if code < n:
delta += 1
elif code == n:
q = delta
k = BASE
while True:
t = max(TMIN, min(TMAX, k - bias))
if q < t:
break
output.append(_digit_to_char(t + (q - t) % (BASE - t)))
q = (q - t) // (BASE - t)
k += BASE
output.append(_digit_to_char(q))
bias = _adapt(delta, h + 1, h == b)
delta = 0
h += 1
delta += 1
n += 1
return "".join(output)
def decode_label(input: str) -> Optional[str]:
"""Punycode-decode a single label (RFC 3492). Returns the decoded label, or
None if the input is malformed (invalid digit, truncated generalized number,
non-ASCII in the basic portion, or arithmetic overflow). Twin of
decodeLabel() in the TS (string | null) and Go (bool form)."""
last_dash = input.rfind("-")
output: List[str] = []
if last_dash >= 0:
for c in input[:last_dash]:
if ord(c) >= 128:
return None # basic portion must be ASCII
output.append(c)
ext = input[last_dash + 1:] if last_dash >= 0 else input
n = INITIAL_N
i = 0
bias = INITIAL_BIAS
pos = 0
while pos < len(ext):
oldi = i
w = 1
k = BASE
while True:
if pos >= len(ext):
return None # truncated generalized number
digit = _char_to_digit(ext[pos])
if digit < 0:
return None # invalid digit
pos += 1
if digit >= MAX_INT // w:
return None # overflow guard
i += digit * w
t = max(TMIN, min(TMAX, k - bias))
if digit < t:
break
w *= BASE - t
k += BASE
bias = _adapt(i - oldi, len(output) + 1, oldi == 0)
out_len = len(output) + 1
n += i // out_len
i %= out_len
output.insert(i, _cp_to_char(n)) # ≡ output.splice(i, 0, …)
i += 1
return "".join(output)
def encode(domain: str) -> str:
"""IDNA toASCII: encode a domain to Punycode ("xn--") form. Lowercases the
whole domain, splits on ".", ACE-encodes any label containing a non-ASCII
code point, leaves ASCII-only labels untouched, and rejoins with ".". Empty
input returns empty. Twin of encode() in the TS/Go."""
if len(domain) == 0:
return ""
lower = domain.lower()
return ".".join(
ACE_PREFIX + encode_label(label) if _has_non_ascii(label) else label
for label in lower.split(".")
)
def decode(domain: str) -> Optional[str]:
"""IDNA toUnicode: decode a Punycode ("xn--") domain back to Unicode. Splits
on ".", decodes any label beginning with "xn--" (case-insensitive), leaves
every other label untouched, and rejoins with ".". Returns None if any
"xn--" label is invalid — the whole domain is rejected, matching IDNA
semantics. Empty input returns "". Twin of decode() in the TS/Go."""
if len(domain) == 0:
return ""
out: List[str] = []
for label in domain.split("."):
if label.lower().startswith(ACE_PREFIX) and len(label) > len(ACE_PREFIX):
decoded = decode_label(label[len(ACE_PREFIX):])
if decoded is None:
return None
out.append(decoded)
else:
out.append(label)
return ".".join(out)
if __name__ == "__main__":
# Showcase vectors — shared with the TS/Go/Rust/PHP/JS twins so every
# implementation is held to one contract.
assert encode("münchen.de") == "xn--mnchen-3ya.de"
assert decode("xn--mnchen-3ya.de") == "münchen.de"
assert encode("Bücher.DE") == "xn--bcher-kva.de" # lowercased first
assert encode_label("café") == "caf-dma"
assert decode_label("caf-dma") == "café"
assert decode("xn--!") is None # invalid digit
assert decode_label("9" * 40) is None # overflow guard
assert decode(encode("café.fr")) == "café.fr"
print("punycode: all showcase vectors passed")
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →