Punycode Converter — TypeScript source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Pure Punycode (RFC 3492) + IDNA2003 toASCII/toUnicode for domain labels.
// Zero deps - the unit-test surface for the Punycode Converter tool.
// RFC 3492 parameters (section 5).
const BASE = 36;
const TMIN = 1;
const TMAX = 26;
const SKEW = 38;
const DAMP = 700;
const INITIAL_BIAS = 72;
const INITIAL_N = 128;
const ACE_PREFIX = 'xn--';
const MAX_INT = Number.MAX_SAFE_INTEGER; // overflow guard for malformed input
/** Bias adaptation (RFC 3492 section 6.1). */
function adapt(delta: number, numpoints: number, firsttime: boolean): number {
let d = firsttime ? Math.floor(delta / DAMP) : Math.floor(delta / 2);
d += Math.floor(d / numpoints);
let k = 0;
while (d > Math.floor(((BASE - TMIN) * TMAX) / 2)) {
d = Math.floor(d / (BASE - TMIN));
k += BASE;
}
return k + Math.floor(((BASE - TMIN + 1) * d) / (d + SKEW));
}
/** Map a digit value (0-35) to its RFC 3492 base-36 character (lowercase). */
function digitToChar(d: number): string {
return d < 26
? String.fromCharCode(97 + d) // a-z
: String.fromCharCode(48 + (d - 26)); // 0-9
}
/** Map a character to its digit value (0-35), case-insensitive, or -1 if invalid. */
function charToDigit(c: string): number {
const code = c.charCodeAt(0);
if (code >= 97 && code <= 122) return code - 97; // a-z
if (code >= 65 && code <= 90) return code - 65; // A-Z
if (code >= 48 && code <= 57) return code - 48 + 26; // 0-9
return -1;
}
/** True if the string contains any non-ASCII code point (>= 128). */
function hasNonAscii(s: string): boolean {
for (let i = 0; i < s.length; i++) {
if (s.charCodeAt(i) >= 128) return true; // surrogates (astral chars) are >= 128
}
return false;
}
/**
* Punycode-encode a single label (RFC 3492). Returns the encoded label with no
* ACE prefix. Basic (ASCII) code points are emitted first, followed by a `-`
* delimiter (only if there was at least one), then the generalized-base-36
* deltas for the non-basic code points.
*/
export function encodeLabel(input: string): string {
// Iterate by code point so astral characters (e.g. emoji, CJK extensions)
// are handled as single elements.
const codePoints: number[] = [];
for (const ch of input) codePoints.push(ch.codePointAt(0)!);
const length = codePoints.length;
const output: string[] = [];
for (const cp of codePoints) {
if (cp < 128) output.push(String.fromCodePoint(cp));
}
const b = output.length;
if (b > 0) output.push('-');
let n = INITIAL_N;
let delta = 0;
let bias = INITIAL_BIAS;
let h = b;
while (h < length) {
// Smallest code point in the input that is >= n.
let m = Infinity;
for (const cp of codePoints) if (cp >= n && cp < m) m = cp;
delta += (m - n) * (h + 1);
n = m;
for (const cp of codePoints) {
if (cp < n) {
delta += 1;
} else if (cp === n) {
let q = delta;
for (let k = BASE; ; k += BASE) {
const t = Math.max(TMIN, Math.min(TMAX, k - bias));
if (q < t) break;
output.push(digitToChar(t + ((q - t) % (BASE - t))));
q = Math.floor((q - t) / (BASE - t));
}
output.push(digitToChar(q));
bias = adapt(delta, h + 1, h === b);
delta = 0;
h += 1;
}
}
delta += 1;
n += 1;
}
return output.join('');
}
/**
* Punycode-decode a single label (RFC 3492). Returns the decoded label, or
* `null` if the input is malformed (invalid digit, truncated generalized
* number, non-ASCII in the basic portion, or arithmetic overflow).
*/
export function decodeLabel(input: string): string | null {
const lastDash = input.lastIndexOf('-');
const output: string[] = [];
if (lastDash >= 0) {
for (let i = 0; i < lastDash; i++) {
if (input.charCodeAt(i) >= 128) return null; // basic portion must be ASCII
output.push(input[i]);
}
}
const ext = lastDash >= 0 ? input.slice(lastDash + 1) : input;
let n = INITIAL_N;
let i = 0;
let bias = INITIAL_BIAS;
let pos = 0;
while (pos < ext.length) {
const oldi = i;
let w = 1;
for (let k = BASE; ; k += BASE) {
if (pos >= ext.length) return null; // truncated generalized number
const digit = charToDigit(ext[pos]);
if (digit < 0) return null; // invalid digit
pos += 1;
if (digit >= MAX_INT / w) return null; // overflow guard
i += digit * w;
const t = Math.max(TMIN, Math.min(TMAX, k - bias));
if (digit < t) break;
w *= BASE - t;
}
bias = adapt(i - oldi, output.length + 1, oldi === 0);
const outLen = output.length + 1;
n += Math.floor(i / outLen);
i %= outLen;
output.splice(i, 0, String.fromCodePoint(n));
i += 1;
}
return output.join('');
}
/**
* IDNA toASCII: encode a domain to Punycode (`xn--`) form.
*
* Lowercases the whole domain, splits on `.`, ACE-encodes (`xn--` + Punycode)
* any label containing a non-ASCII code point, leaves ASCII-only labels
* untouched, and rejoins with `.`. Empty input returns empty.
*/
export function encode(domain: string): string {
if (domain.length === 0) return '';
const lower = domain.toLowerCase();
return lower
.split('.')
.map((label) => (hasNonAscii(label) ? ACE_PREFIX + encodeLabel(label) : label))
.join('.');
}
/**
* IDNA toUnicode: decode a Punycode (`xn--`) domain back to Unicode.
*
* Splits on `.`, decodes any label beginning with `xn--` (case-insensitive,
* prefix detected on the lowercased label), leaves every other label
* untouched, and rejoins with `.`. Returns `null` if any `xn--` label is
* invalid - the whole domain is rejected, matching IDNA semantics. Empty input
* returns empty.
*/
export function decode(domain: string): string | null {
if (domain.length === 0) return '';
const out: string[] = [];
for (const label of domain.split('.')) {
if (label.toLowerCase().startsWith(ACE_PREFIX) && label.length > ACE_PREFIX.length) {
const decoded = decodeLabel(label.slice(ACE_PREFIX.length));
if (decoded === null) return null;
out.push(decoded);
} else {
out.push(label);
}
}
return out.join('.');
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →