Skip to content

Punycode Converter — Python source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""punycode — RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.

Language: Python (3.9+, standard library only)
Source:   CosmoDev polyglot showcase port of the Punycode tool, ported from
          src/lib/punycode.ts (canonical TypeScript) and cli/punycode/punycode.go
          (the live Go CLI twin — the two are kept in lock-step).
License:  display source — part of CosmoDev's polyglot tool pages.

Design goals:
  - Pure + deterministic; never raises (decode returns Optional[str], None on
    malformed input).
  - Functionally equivalent to the Go/TS reference: same inputs -> same outputs.
  - Self-contained: stdlib only (no pip packages).

Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
helpers. encode_label/decode_label operate on a single label (no ACE prefix);
encode/decode wrap them with the "xn--" prefixing and "." splitting of a full
domain. Python strings are Unicode code-point sequences (≡ Go runes ≡ the TS
code-point iteration), so bias adaptation, generalized base-36 digits, and the
2^53-1 overflow guard map 1:1. Python's // is floor division; every operand
here is non-negative, so it matches Go's truncated integer division exactly.
"""

from __future__ import annotations

from typing import List, Optional

__all__ = ["encode", "decode", "encode_label", "decode_label"]

# RFC 3492 parameters (section 5), matching the TS/Go constants.
BASE = 36
TMIN = 1
TMAX = 26
SKEW = 38
DAMP = 700
INITIAL_BIAS = 72
INITIAL_N = 128
ACE_PREFIX = "xn--"
# Mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for malformed
# decode input. Makes pathological generalized numbers (e.g. 40 nines) return
# None instead of running away.
MAX_INT = (1 << 53) - 1


def _adapt(delta: int, numpoints: int, firsttime: bool) -> int:
    """Bias adaptation (RFC 3492 section 6.1). delta/numpoints are always
    non-negative, so // matches the reference integer division exactly."""
    d = (delta // DAMP) if firsttime else (delta // 2)
    d += d // numpoints
    k = 0
    while d > ((BASE - TMIN) * TMAX) // 2:
        d //= BASE - TMIN
        k += BASE
    return k + (BASE - TMIN + 1) * d // (d + SKEW)


def _digit_to_char(d: int) -> str:
    """Map a digit value (0–35) to its RFC 3492 base-36 character (lowercase
    a–z for 0–25, 0–9 for 26–35). d is always in [0,35] on the encode path."""
    return chr(ord("a") + d) if d < 26 else chr(ord("0") + (d - 26))


def _char_to_digit(c: str) -> int:
    """Map a character to its digit value (0–35), case-insensitive, or -1 if it
    is not a valid base-36 digit. Any non-ASCII char yields -1, mirroring the
    reference behavior of rejecting non-ASCII in the extension portion."""
    code = ord(c)
    if 0x61 <= code <= 0x7A:  # a–z
        return code - 0x61
    if 0x41 <= code <= 0x5A:  # A–Z
        return code - 0x41
    if 0x30 <= code <= 0x39:  # 0–9
        return code - 0x30 + 26
    return -1


def _has_non_ascii(s: str) -> bool:
    """True if the string contains any non-ASCII code point (>= 128)."""
    return any(ord(c) >= 128 for c in s)


def _cp_to_char(n: int) -> str:
    """Safely turn a decoded code point into a character. Mirrors Go's rune(n)
    (which emits U+FFFD for an out-of-range or surrogate value) rather than
    raising."""
    return chr(n) if 0 <= n <= 0x10FFFF else "�"


def encode_label(input: str) -> str:
    """Punycode-encode a single label (RFC 3492) and return the encoded label
    with no ACE prefix. Basic (ASCII) code points are emitted first, followed by
    a '-' delimiter (only if there was at least one), then the generalized
    base-36 deltas for the non-basic code points. Twin of encodeLabel() in the
    TS/Go."""
    code_points = list(input)  # iterate by code point (astral chars are single)
    length = len(code_points)

    output: List[str] = [c for c in code_points if ord(c) < 128]
    b = len(output)
    if b > 0:
        output.append("-")

    n = INITIAL_N
    delta = 0
    bias = INITIAL_BIAS
    h = b

    while h < length:
        # Smallest code point in the input that is >= n. The while guard ensures
        # at least one such code point remains, so min() never gets an empty
        # iterable. Single-char comparison is by code point (≡ by ord).
        m = min(c for c in code_points if ord(c) >= n)
        delta += (ord(m) - n) * (h + 1)
        n = ord(m)
        for c in code_points:
            code = ord(c)
            if code < n:
                delta += 1
            elif code == n:
                q = delta
                k = BASE
                while True:
                    t = max(TMIN, min(TMAX, k - bias))
                    if q < t:
                        break
                    output.append(_digit_to_char(t + (q - t) % (BASE - t)))
                    q = (q - t) // (BASE - t)
                    k += BASE
                output.append(_digit_to_char(q))
                bias = _adapt(delta, h + 1, h == b)
                delta = 0
                h += 1
        delta += 1
        n += 1

    return "".join(output)


def decode_label(input: str) -> Optional[str]:
    """Punycode-decode a single label (RFC 3492). Returns the decoded label, or
    None if the input is malformed (invalid digit, truncated generalized number,
    non-ASCII in the basic portion, or arithmetic overflow). Twin of
    decodeLabel() in the TS (string | null) and Go (bool form)."""
    last_dash = input.rfind("-")
    output: List[str] = []
    if last_dash >= 0:
        for c in input[:last_dash]:
            if ord(c) >= 128:
                return None  # basic portion must be ASCII
            output.append(c)
    ext = input[last_dash + 1:] if last_dash >= 0 else input

    n = INITIAL_N
    i = 0
    bias = INITIAL_BIAS
    pos = 0

    while pos < len(ext):
        oldi = i
        w = 1
        k = BASE
        while True:
            if pos >= len(ext):
                return None  # truncated generalized number
            digit = _char_to_digit(ext[pos])
            if digit < 0:
                return None  # invalid digit
            pos += 1
            if digit >= MAX_INT // w:
                return None  # overflow guard
            i += digit * w
            t = max(TMIN, min(TMAX, k - bias))
            if digit < t:
                break
            w *= BASE - t
            k += BASE
        bias = _adapt(i - oldi, len(output) + 1, oldi == 0)
        out_len = len(output) + 1
        n += i // out_len
        i %= out_len
        output.insert(i, _cp_to_char(n))  # ≡ output.splice(i, 0, …)
        i += 1

    return "".join(output)


def encode(domain: str) -> str:
    """IDNA toASCII: encode a domain to Punycode ("xn--") form. Lowercases the
    whole domain, splits on ".", ACE-encodes any label containing a non-ASCII
    code point, leaves ASCII-only labels untouched, and rejoins with ".". Empty
    input returns empty. Twin of encode() in the TS/Go."""
    if len(domain) == 0:
        return ""
    lower = domain.lower()
    return ".".join(
        ACE_PREFIX + encode_label(label) if _has_non_ascii(label) else label
        for label in lower.split(".")
    )


def decode(domain: str) -> Optional[str]:
    """IDNA toUnicode: decode a Punycode ("xn--") domain back to Unicode. Splits
    on ".", decodes any label beginning with "xn--" (case-insensitive), leaves
    every other label untouched, and rejoins with ".". Returns None if any
    "xn--" label is invalid — the whole domain is rejected, matching IDNA
    semantics. Empty input returns "". Twin of decode() in the TS/Go."""
    if len(domain) == 0:
        return ""
    out: List[str] = []
    for label in domain.split("."):
        if label.lower().startswith(ACE_PREFIX) and len(label) > len(ACE_PREFIX):
            decoded = decode_label(label[len(ACE_PREFIX):])
            if decoded is None:
                return None
            out.append(decoded)
        else:
            out.append(label)
    return ".".join(out)


if __name__ == "__main__":
    # Showcase vectors — shared with the TS/Go/Rust/PHP/JS twins so every
    # implementation is held to one contract.
    assert encode("münchen.de") == "xn--mnchen-3ya.de"
    assert decode("xn--mnchen-3ya.de") == "münchen.de"
    assert encode("Bücher.DE") == "xn--bcher-kva.de"  # lowercased first
    assert encode_label("café") == "caf-dma"
    assert decode_label("caf-dma") == "café"
    assert decode("xn--!") is None              # invalid digit
    assert decode_label("9" * 40) is None       # overflow guard
    assert decode(encode("café.fr")) == "café.fr"
    print("punycode: all showcase vectors passed")

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →