Skip to content

Punycode Converter — Rust source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! punycode — RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source:   CosmoDev polyglot showcase port of the Punycode tool, ported from
//!           src/lib/punycode.ts (canonical TypeScript) and cli/punycode/punycode.go
//!           (the live Go CLI twin — the two are kept in lock-step).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics (decode returns Option<String>).
//!   - Functionally equivalent to the Go/TS reference: same inputs -> same outputs.
//!   - Self-contained: std only (no crates.io dependencies).
//!
//! Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
//! helpers. encode_label/decode_label operate on a single label (no ACE prefix);
//! encode/decode wrap them with the "xn--" prefixing and "." splitting of a full
//! domain. Code points are iterated as Rust chars (≡ Go runes ≡ the TS code-point
//! iteration); bias adaptation, generalized base-36 digits, and the 2^53-1
//! overflow guard all map 1:1.

/// RFC 3492 parameters (section 5), matching the TS/Go constants.
const BASE: i64 = 36;
const TMIN: i64 = 1;
const TMAX: i64 = 26;
const SKEW: i64 = 38;
const DAMP: i64 = 700;
const INITIAL_BIAS: i64 = 72;
const INITIAL_N: i64 = 128;
const ACE_PREFIX: &str = "xn--";
/// Mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for malformed
/// decode input. Makes pathological generalized numbers (e.g. 40 nines) return
/// None instead of running away.
const MAX_INT: i64 = (1i64 << 53) - 1;

/// Bias adaptation (RFC 3492 section 6.1). delta and numpoints are always
/// non-negative, so Rust's integer division matches the TS Math.floor divisions
/// and Go's integer division exactly.
fn adapt(delta: i64, numpoints: i64, firsttime: bool) -> i64 {
    let mut d = if firsttime { delta / DAMP } else { delta / 2 };
    d += d / numpoints;
    let mut k = 0i64;
    while d > ((BASE - TMIN) * TMAX) / 2 {
        d /= BASE - TMIN;
        k += BASE;
    }
    k + (BASE - TMIN + 1) * d / (d + SKEW)
}

/// Map a digit value (0–35) to its RFC 3492 base-36 character (lowercase a–z for
/// 0–25, 0–9 for 26–35). `d` is always in [0,35] on the encode path, so the
/// unwrap never fires.
fn digit_to_char(d: i64) -> char {
    if d < 26 {
        char::from_u32('a' as u32 + d as u32).unwrap() // a–z
    } else {
        char::from_u32('0' as u32 + (d - 26) as u32).unwrap() // 0–9
    }
}

/// Map a character to its digit value (0–35), case-insensitive, or -1 if it is
/// not a valid base-36 digit. Any non-ASCII char matches no case and yields -1,
/// mirroring the reference behavior of rejecting non-ASCII in the extension
/// portion.
fn char_to_digit(c: char) -> i64 {
    let v = c as u32;
    if (0x61..=0x7A).contains(&v) {
        (v - 0x61) as i64 // a–z
    } else if (0x41..=0x5A).contains(&v) {
        (v - 0x41) as i64 // A–Z
    } else if (0x30..=0x39).contains(&v) {
        (v - 0x30 + 26) as i64 // 0–9
    } else {
        -1
    }
}

/// Reports whether `s` contains any non-ASCII code point (>= 128). Iterating
/// chars is equivalent to the reference's unit/rune scan: any non-ASCII
/// character has a code point >= 128.
fn has_non_ascii(s: &str) -> bool {
    s.chars().any(|c| (c as u32) >= 128)
}

/// Safely turn a decoded code point into a Rust char. Mirrors Go's `rune(n)`
/// (which emits U+FFFD for an out-of-range or surrogate value) rather than
/// panicking.
fn cp_to_char(n: i64) -> char {
    char::from_u32(n as u32).unwrap_or('\u{FFFD}')
}

/// Punycode-encode a single label (RFC 3492) and return the encoded label with
/// no ACE prefix. Basic (ASCII) code points are emitted first, followed by a
/// `-` delimiter (only if there was at least one), then the generalized base-36
/// deltas for the non-basic code points. It is the Rust twin of encodeLabel()
/// in src/lib/punycode.ts and cli/punycode/punycode.go.
pub fn encode_label(input: &str) -> String {
    // Iterate by char (code point) so astral characters (emoji, CJK extensions)
    // are handled as single elements — same as Go's rune / TS code-point range.
    let code_points: Vec<u32> = input.chars().map(|c| c as u32).collect();
    let length = code_points.len() as i64;

    let mut output: Vec<char> = Vec::new();
    for &cp in &code_points {
        if cp < 128 {
            output.push(char::from_u32(cp).unwrap());
        }
    }
    let b = output.len() as i64;
    if b > 0 {
        output.push('-');
    }

    let mut n = INITIAL_N;
    let mut delta = 0i64;
    let mut bias = INITIAL_BIAS;
    let mut h = b;

    while h < length {
        // Smallest code point in the input that is >= n.
        let mut m: i64 = -1;
        for &cp in &code_points {
            let c = cp as i64;
            if c >= n && (m == -1 || c < m) {
                m = c;
            }
        }
        delta += (m - n) * (h + 1);
        n = m;
        for &cp in &code_points {
            let c = cp as i64;
            if c < n {
                delta += 1;
            } else if c == n {
                let mut q = delta;
                let mut k = BASE;
                loop {
                    let mut t = k - bias;
                    if t < TMIN {
                        t = TMIN;
                    } else if t > TMAX {
                        t = TMAX;
                    } // t = max(tmin, min(tmax, k - bias))
                    if q < t {
                        break;
                    }
                    output.push(digit_to_char(t + (q - t) % (BASE - t)));
                    q = (q - t) / (BASE - t);
                    k += BASE;
                }
                output.push(digit_to_char(q));
                bias = adapt(delta, h + 1, h == b);
                delta = 0;
                h += 1;
            }
        }
        delta += 1;
        n += 1;
    }

    output.into_iter().collect()
}

/// Punycode-decode a single label (RFC 3492). Returns the decoded label, or
/// `None` if the input is malformed (invalid digit, truncated generalized
/// number, non-ASCII in the basic portion, or arithmetic overflow). It is the
/// Rust twin of decodeLabel() in src/lib/punycode.ts (which returns
/// `string | null`) and cli/punycode/punycode.go (the bool form).
pub fn decode_label(input: &str) -> Option<String> {
    // last_dash is a byte index; since '-' is ASCII it coincides with the
    // code-point index the TS twin uses.
    let last_dash = input.as_bytes().iter().rposition(|&b| b == b'-');
    let mut output: Vec<char> = Vec::new();
    if let Some(ld) = last_dash {
        // Basic portion (before the last dash) must be pure ASCII.
        for &b in &input.as_bytes()[..ld] {
            if b >= 128 {
                return None;
            }
            output.push(b as char);
        }
    }
    let ext: String = match last_dash {
        Some(ld) => input[ld + 1..].to_string(),
        None => input.to_string(),
    };

    let mut n = INITIAL_N;
    let mut i = 0i64;
    let mut bias = INITIAL_BIAS;
    let ext_chars: Vec<char> = ext.chars().collect();
    let mut pos = 0usize;

    while pos < ext_chars.len() {
        let oldi = i;
        let mut w = 1i64;
        let mut k = BASE;
        loop {
            if pos >= ext_chars.len() {
                return None; // truncated generalized number
            }
            let digit = char_to_digit(ext_chars[pos]);
            if digit < 0 {
                return None; // invalid digit
            }
            pos += 1;
            if digit >= MAX_INT / w {
                return None; // overflow guard
            }
            i += digit * w;
            let mut t = k - bias;
            if t < TMIN {
                t = TMIN;
            } else if t > TMAX {
                t = TMAX;
            } // t = max(tmin, min(tmax, k - bias))
            if digit < t {
                break;
            }
            w *= BASE - t;
            k += BASE;
        }
        bias = adapt(i - oldi, output.len() as i64 + 1, oldi == 0);
        let out_len = output.len() as i64 + 1;
        n += i / out_len;
        i %= out_len;
        // Insert the decoded code point n at index i (≡ output.splice(i, 0, …)).
        output.insert(i as usize, cp_to_char(n));
        i += 1;
    }

    Some(output.into_iter().collect())
}

/// IDNA toASCII: encode a domain to Punycode ("xn--") form. Lowercases the whole
/// domain, splits on ".", ACE-encodes ("xn--" + Punycode) any label containing a
/// non-ASCII code point, leaves ASCII-only labels untouched, and rejoins with
/// ".". Empty input returns empty. It is the Rust twin of encode() in the TS/Go.
pub fn encode(domain: &str) -> String {
    if domain.is_empty() {
        return String::new();
    }
    let lower = domain.to_lowercase();
    lower
        .split('.')
        .map(|label| {
            if has_non_ascii(label) {
                let mut s = String::from(ACE_PREFIX);
                s.push_str(&encode_label(label));
                s
            } else {
                label.to_string()
            }
        })
        .collect::<Vec<_>>()
        .join(".")
}

/// IDNA toUnicode: decode a Punycode ("xn--") domain back to Unicode. Splits on
/// ".", decodes any label beginning with "xn--" (case-insensitive, detected on
/// the lowercased label), leaves every other label untouched, and rejoins with
/// ".". Returns `None` if any "xn--" label is invalid — the whole domain is
/// rejected, matching IDNA semantics. Empty input returns `Some("")`. It is the
/// Rust twin of decode() in the TS/Go.
pub fn decode(domain: &str) -> Option<String> {
    if domain.is_empty() {
        return Some(String::new());
    }
    let mut out: Vec<String> = Vec::new();
    for label in domain.split('.') {
        let lower = label.to_lowercase();
        if lower.starts_with(ACE_PREFIX) && label.len() > ACE_PREFIX.len() {
            match decode_label(&label[ACE_PREFIX.len()..]) {
                Some(decoded) => out.push(decoded),
                None => return None,
            }
        } else {
            out.push(label.to_string());
        }
    }
    Some(out.join("."))
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
    use super::*;

    // Showcase vectors — shared with the TS/Go/PHP/Python/JS twins so every
    // implementation is held to one contract.

    #[test]
    fn encode_known_idn() {
        assert_eq!(encode("münchen.de"), "xn--mnchen-3ya.de");
        assert_eq!(encode("Bücher.DE"), "xn--bcher-kva.de"); // lowercased first
    }

    #[test]
    fn decode_known_ace() {
        assert_eq!(decode("xn--mnchen-3ya.de").as_deref(), Some("münchen.de"));
    }

    #[test]
    fn label_round_trip() {
        assert_eq!(encode_label("café"), "caf-dma");
        assert_eq!(decode_label("caf-dma").as_deref(), Some("café"));
    }

    #[test]
    fn round_trip_domain() {
        assert_eq!(decode(&encode("café.fr")).as_deref(), Some("café.fr"));
    }

    #[test]
    fn invalid_returns_none() {
        assert_eq!(decode("xn--!"), None);                       // invalid digit
        assert_eq!(decode_label(&"9".repeat(40)), None);         // overflow guard
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →