Skip to content

Punycode Converter — TypeScript source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Pure Punycode (RFC 3492) + IDNA2003 toASCII/toUnicode for domain labels.
// Zero deps - the unit-test surface for the Punycode Converter tool.

// RFC 3492 parameters (section 5).
const BASE = 36;
const TMIN = 1;
const TMAX = 26;
const SKEW = 38;
const DAMP = 700;
const INITIAL_BIAS = 72;
const INITIAL_N = 128;
const ACE_PREFIX = 'xn--';
const MAX_INT = Number.MAX_SAFE_INTEGER; // overflow guard for malformed input

/** Bias adaptation (RFC 3492 section 6.1). */
function adapt(delta: number, numpoints: number, firsttime: boolean): number {
  let d = firsttime ? Math.floor(delta / DAMP) : Math.floor(delta / 2);
  d += Math.floor(d / numpoints);
  let k = 0;
  while (d > Math.floor(((BASE - TMIN) * TMAX) / 2)) {
    d = Math.floor(d / (BASE - TMIN));
    k += BASE;
  }
  return k + Math.floor(((BASE - TMIN + 1) * d) / (d + SKEW));
}

/** Map a digit value (0-35) to its RFC 3492 base-36 character (lowercase). */
function digitToChar(d: number): string {
  return d < 26
    ? String.fromCharCode(97 + d)            // a-z
    : String.fromCharCode(48 + (d - 26));    // 0-9
}

/** Map a character to its digit value (0-35), case-insensitive, or -1 if invalid. */
function charToDigit(c: string): number {
  const code = c.charCodeAt(0);
  if (code >= 97 && code <= 122) return code - 97;      // a-z
  if (code >= 65 && code <= 90) return code - 65;       // A-Z
  if (code >= 48 && code <= 57) return code - 48 + 26;  // 0-9
  return -1;
}

/** True if the string contains any non-ASCII code point (>= 128). */
function hasNonAscii(s: string): boolean {
  for (let i = 0; i < s.length; i++) {
    if (s.charCodeAt(i) >= 128) return true; // surrogates (astral chars) are >= 128
  }
  return false;
}

/**
 * Punycode-encode a single label (RFC 3492). Returns the encoded label with no
 * ACE prefix. Basic (ASCII) code points are emitted first, followed by a `-`
 * delimiter (only if there was at least one), then the generalized-base-36
 * deltas for the non-basic code points.
 */
export function encodeLabel(input: string): string {
  // Iterate by code point so astral characters (e.g. emoji, CJK extensions)
  // are handled as single elements.
  const codePoints: number[] = [];
  for (const ch of input) codePoints.push(ch.codePointAt(0)!);
  const length = codePoints.length;

  const output: string[] = [];
  for (const cp of codePoints) {
    if (cp < 128) output.push(String.fromCodePoint(cp));
  }
  const b = output.length;
  if (b > 0) output.push('-');

  let n = INITIAL_N;
  let delta = 0;
  let bias = INITIAL_BIAS;
  let h = b;

  while (h < length) {
    // Smallest code point in the input that is >= n.
    let m = Infinity;
    for (const cp of codePoints) if (cp >= n && cp < m) m = cp;
    delta += (m - n) * (h + 1);
    n = m;
    for (const cp of codePoints) {
      if (cp < n) {
        delta += 1;
      } else if (cp === n) {
        let q = delta;
        for (let k = BASE; ; k += BASE) {
          const t = Math.max(TMIN, Math.min(TMAX, k - bias));
          if (q < t) break;
          output.push(digitToChar(t + ((q - t) % (BASE - t))));
          q = Math.floor((q - t) / (BASE - t));
        }
        output.push(digitToChar(q));
        bias = adapt(delta, h + 1, h === b);
        delta = 0;
        h += 1;
      }
    }
    delta += 1;
    n += 1;
  }

  return output.join('');
}

/**
 * Punycode-decode a single label (RFC 3492). Returns the decoded label, or
 * `null` if the input is malformed (invalid digit, truncated generalized
 * number, non-ASCII in the basic portion, or arithmetic overflow).
 */
export function decodeLabel(input: string): string | null {
  const lastDash = input.lastIndexOf('-');
  const output: string[] = [];
  if (lastDash >= 0) {
    for (let i = 0; i < lastDash; i++) {
      if (input.charCodeAt(i) >= 128) return null; // basic portion must be ASCII
      output.push(input[i]);
    }
  }
  const ext = lastDash >= 0 ? input.slice(lastDash + 1) : input;

  let n = INITIAL_N;
  let i = 0;
  let bias = INITIAL_BIAS;
  let pos = 0;

  while (pos < ext.length) {
    const oldi = i;
    let w = 1;
    for (let k = BASE; ; k += BASE) {
      if (pos >= ext.length) return null;        // truncated generalized number
      const digit = charToDigit(ext[pos]);
      if (digit < 0) return null;                 // invalid digit
      pos += 1;
      if (digit >= MAX_INT / w) return null;      // overflow guard
      i += digit * w;
      const t = Math.max(TMIN, Math.min(TMAX, k - bias));
      if (digit < t) break;
      w *= BASE - t;
    }
    bias = adapt(i - oldi, output.length + 1, oldi === 0);
    const outLen = output.length + 1;
    n += Math.floor(i / outLen);
    i %= outLen;
    output.splice(i, 0, String.fromCodePoint(n));
    i += 1;
  }

  return output.join('');
}

/**
 * IDNA toASCII: encode a domain to Punycode (`xn--`) form.
 *
 * Lowercases the whole domain, splits on `.`, ACE-encodes (`xn--` + Punycode)
 * any label containing a non-ASCII code point, leaves ASCII-only labels
 * untouched, and rejoins with `.`. Empty input returns empty.
 */
export function encode(domain: string): string {
  if (domain.length === 0) return '';
  const lower = domain.toLowerCase();
  return lower
    .split('.')
    .map((label) => (hasNonAscii(label) ? ACE_PREFIX + encodeLabel(label) : label))
    .join('.');
}

/**
 * IDNA toUnicode: decode a Punycode (`xn--`) domain back to Unicode.
 *
 * Splits on `.`, decodes any label beginning with `xn--` (case-insensitive,
 * prefix detected on the lowercased label), leaves every other label
 * untouched, and rejoins with `.`. Returns `null` if any `xn--` label is
 * invalid - the whole domain is rejected, matching IDNA semantics. Empty input
 * returns empty.
 */
export function decode(domain: string): string | null {
  if (domain.length === 0) return '';
  const out: string[] = [];
  for (const label of domain.split('.')) {
    if (label.toLowerCase().startsWith(ACE_PREFIX) && label.length > ACE_PREFIX.length) {
      const decoded = decodeLabel(label.slice(ACE_PREFIX.length));
      if (decoded === null) return null;
      out.push(decoded);
    } else {
      out.push(label);
    }
  }
  return out.join('.');
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →