Punycode Converter — Rust source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! punycode — RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the Punycode tool, ported from
//! src/lib/punycode.ts (canonical TypeScript) and cli/punycode/punycode.go
//! (the live Go CLI twin — the two are kept in lock-step).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (decode returns Option<String>).
//! - Functionally equivalent to the Go/TS reference: same inputs -> same outputs.
//! - Self-contained: std only (no crates.io dependencies).
//!
//! Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
//! helpers. encode_label/decode_label operate on a single label (no ACE prefix);
//! encode/decode wrap them with the "xn--" prefixing and "." splitting of a full
//! domain. Code points are iterated as Rust chars (≡ Go runes ≡ the TS code-point
//! iteration); bias adaptation, generalized base-36 digits, and the 2^53-1
//! overflow guard all map 1:1.
/// RFC 3492 parameters (section 5), matching the TS/Go constants.
const BASE: i64 = 36;
const TMIN: i64 = 1;
const TMAX: i64 = 26;
const SKEW: i64 = 38;
const DAMP: i64 = 700;
const INITIAL_BIAS: i64 = 72;
const INITIAL_N: i64 = 128;
const ACE_PREFIX: &str = "xn--";
/// Mirrors Number.MAX_SAFE_INTEGER (2^53-1): the overflow guard for malformed
/// decode input. Makes pathological generalized numbers (e.g. 40 nines) return
/// None instead of running away.
const MAX_INT: i64 = (1i64 << 53) - 1;
/// Bias adaptation (RFC 3492 section 6.1). delta and numpoints are always
/// non-negative, so Rust's integer division matches the TS Math.floor divisions
/// and Go's integer division exactly.
fn adapt(delta: i64, numpoints: i64, firsttime: bool) -> i64 {
let mut d = if firsttime { delta / DAMP } else { delta / 2 };
d += d / numpoints;
let mut k = 0i64;
while d > ((BASE - TMIN) * TMAX) / 2 {
d /= BASE - TMIN;
k += BASE;
}
k + (BASE - TMIN + 1) * d / (d + SKEW)
}
/// Map a digit value (0–35) to its RFC 3492 base-36 character (lowercase a–z for
/// 0–25, 0–9 for 26–35). `d` is always in [0,35] on the encode path, so the
/// unwrap never fires.
fn digit_to_char(d: i64) -> char {
if d < 26 {
char::from_u32('a' as u32 + d as u32).unwrap() // a–z
} else {
char::from_u32('0' as u32 + (d - 26) as u32).unwrap() // 0–9
}
}
/// Map a character to its digit value (0–35), case-insensitive, or -1 if it is
/// not a valid base-36 digit. Any non-ASCII char matches no case and yields -1,
/// mirroring the reference behavior of rejecting non-ASCII in the extension
/// portion.
fn char_to_digit(c: char) -> i64 {
let v = c as u32;
if (0x61..=0x7A).contains(&v) {
(v - 0x61) as i64 // a–z
} else if (0x41..=0x5A).contains(&v) {
(v - 0x41) as i64 // A–Z
} else if (0x30..=0x39).contains(&v) {
(v - 0x30 + 26) as i64 // 0–9
} else {
-1
}
}
/// Reports whether `s` contains any non-ASCII code point (>= 128). Iterating
/// chars is equivalent to the reference's unit/rune scan: any non-ASCII
/// character has a code point >= 128.
fn has_non_ascii(s: &str) -> bool {
s.chars().any(|c| (c as u32) >= 128)
}
/// Safely turn a decoded code point into a Rust char. Mirrors Go's `rune(n)`
/// (which emits U+FFFD for an out-of-range or surrogate value) rather than
/// panicking.
fn cp_to_char(n: i64) -> char {
char::from_u32(n as u32).unwrap_or('\u{FFFD}')
}
/// Punycode-encode a single label (RFC 3492) and return the encoded label with
/// no ACE prefix. Basic (ASCII) code points are emitted first, followed by a
/// `-` delimiter (only if there was at least one), then the generalized base-36
/// deltas for the non-basic code points. It is the Rust twin of encodeLabel()
/// in src/lib/punycode.ts and cli/punycode/punycode.go.
pub fn encode_label(input: &str) -> String {
// Iterate by char (code point) so astral characters (emoji, CJK extensions)
// are handled as single elements — same as Go's rune / TS code-point range.
let code_points: Vec<u32> = input.chars().map(|c| c as u32).collect();
let length = code_points.len() as i64;
let mut output: Vec<char> = Vec::new();
for &cp in &code_points {
if cp < 128 {
output.push(char::from_u32(cp).unwrap());
}
}
let b = output.len() as i64;
if b > 0 {
output.push('-');
}
let mut n = INITIAL_N;
let mut delta = 0i64;
let mut bias = INITIAL_BIAS;
let mut h = b;
while h < length {
// Smallest code point in the input that is >= n.
let mut m: i64 = -1;
for &cp in &code_points {
let c = cp as i64;
if c >= n && (m == -1 || c < m) {
m = c;
}
}
delta += (m - n) * (h + 1);
n = m;
for &cp in &code_points {
let c = cp as i64;
if c < n {
delta += 1;
} else if c == n {
let mut q = delta;
let mut k = BASE;
loop {
let mut t = k - bias;
if t < TMIN {
t = TMIN;
} else if t > TMAX {
t = TMAX;
} // t = max(tmin, min(tmax, k - bias))
if q < t {
break;
}
output.push(digit_to_char(t + (q - t) % (BASE - t)));
q = (q - t) / (BASE - t);
k += BASE;
}
output.push(digit_to_char(q));
bias = adapt(delta, h + 1, h == b);
delta = 0;
h += 1;
}
}
delta += 1;
n += 1;
}
output.into_iter().collect()
}
/// Punycode-decode a single label (RFC 3492). Returns the decoded label, or
/// `None` if the input is malformed (invalid digit, truncated generalized
/// number, non-ASCII in the basic portion, or arithmetic overflow). It is the
/// Rust twin of decodeLabel() in src/lib/punycode.ts (which returns
/// `string | null`) and cli/punycode/punycode.go (the bool form).
pub fn decode_label(input: &str) -> Option<String> {
// last_dash is a byte index; since '-' is ASCII it coincides with the
// code-point index the TS twin uses.
let last_dash = input.as_bytes().iter().rposition(|&b| b == b'-');
let mut output: Vec<char> = Vec::new();
if let Some(ld) = last_dash {
// Basic portion (before the last dash) must be pure ASCII.
for &b in &input.as_bytes()[..ld] {
if b >= 128 {
return None;
}
output.push(b as char);
}
}
let ext: String = match last_dash {
Some(ld) => input[ld + 1..].to_string(),
None => input.to_string(),
};
let mut n = INITIAL_N;
let mut i = 0i64;
let mut bias = INITIAL_BIAS;
let ext_chars: Vec<char> = ext.chars().collect();
let mut pos = 0usize;
while pos < ext_chars.len() {
let oldi = i;
let mut w = 1i64;
let mut k = BASE;
loop {
if pos >= ext_chars.len() {
return None; // truncated generalized number
}
let digit = char_to_digit(ext_chars[pos]);
if digit < 0 {
return None; // invalid digit
}
pos += 1;
if digit >= MAX_INT / w {
return None; // overflow guard
}
i += digit * w;
let mut t = k - bias;
if t < TMIN {
t = TMIN;
} else if t > TMAX {
t = TMAX;
} // t = max(tmin, min(tmax, k - bias))
if digit < t {
break;
}
w *= BASE - t;
k += BASE;
}
bias = adapt(i - oldi, output.len() as i64 + 1, oldi == 0);
let out_len = output.len() as i64 + 1;
n += i / out_len;
i %= out_len;
// Insert the decoded code point n at index i (≡ output.splice(i, 0, …)).
output.insert(i as usize, cp_to_char(n));
i += 1;
}
Some(output.into_iter().collect())
}
/// IDNA toASCII: encode a domain to Punycode ("xn--") form. Lowercases the whole
/// domain, splits on ".", ACE-encodes ("xn--" + Punycode) any label containing a
/// non-ASCII code point, leaves ASCII-only labels untouched, and rejoins with
/// ".". Empty input returns empty. It is the Rust twin of encode() in the TS/Go.
pub fn encode(domain: &str) -> String {
if domain.is_empty() {
return String::new();
}
let lower = domain.to_lowercase();
lower
.split('.')
.map(|label| {
if has_non_ascii(label) {
let mut s = String::from(ACE_PREFIX);
s.push_str(&encode_label(label));
s
} else {
label.to_string()
}
})
.collect::<Vec<_>>()
.join(".")
}
/// IDNA toUnicode: decode a Punycode ("xn--") domain back to Unicode. Splits on
/// ".", decodes any label beginning with "xn--" (case-insensitive, detected on
/// the lowercased label), leaves every other label untouched, and rejoins with
/// ".". Returns `None` if any "xn--" label is invalid — the whole domain is
/// rejected, matching IDNA semantics. Empty input returns `Some("")`. It is the
/// Rust twin of decode() in the TS/Go.
pub fn decode(domain: &str) -> Option<String> {
if domain.is_empty() {
return Some(String::new());
}
let mut out: Vec<String> = Vec::new();
for label in domain.split('.') {
let lower = label.to_lowercase();
if lower.starts_with(ACE_PREFIX) && label.len() > ACE_PREFIX.len() {
match decode_label(&label[ACE_PREFIX.len()..]) {
Some(decoded) => out.push(decoded),
None => return None,
}
} else {
out.push(label.to_string());
}
}
Some(out.join("."))
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
use super::*;
// Showcase vectors — shared with the TS/Go/PHP/Python/JS twins so every
// implementation is held to one contract.
#[test]
fn encode_known_idn() {
assert_eq!(encode("münchen.de"), "xn--mnchen-3ya.de");
assert_eq!(encode("Bücher.DE"), "xn--bcher-kva.de"); // lowercased first
}
#[test]
fn decode_known_ace() {
assert_eq!(decode("xn--mnchen-3ya.de").as_deref(), Some("münchen.de"));
}
#[test]
fn label_round_trip() {
assert_eq!(encode_label("café"), "caf-dma");
assert_eq!(decode_label("caf-dma").as_deref(), Some("café"));
}
#[test]
fn round_trip_domain() {
assert_eq!(decode(&encode("café.fr")).as_deref(), Some("café.fr"));
}
#[test]
fn invalid_returns_none() {
assert_eq!(decode("xn--!"), None); // invalid digit
assert_eq!(decode_label(&"9".repeat(40)), None); // overflow guard
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →