Skip to content

Punycode Converter — JavaScript source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
 *
 * Language:   JavaScript (ES2020+, runs unmodified in Node 16+ and modern browsers)
 * Source:     CosmoDev polyglot showcase port of the Punycode tool, ported from
 *             src/lib/punycode.ts (the canonical TypeScript - this is its
 *             plain-JS twin) and cli/punycode/punycode.go (the live Go CLI twin).
 * License:    display source - part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; never throws (decode returns string | null).
 *   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
 *   - Self-contained: stdlib only (no npm dependencies).
 *
 * Implements RFC 3492 (Punycode) plus the IDNA2003 toASCII/toUnicode label
 * helpers. encodeLabel/decodeLabel operate on a single label (no ACE prefix);
 * encode/decode wrap them with the "xn--" prefixing and "." splitting of a full
 * domain. `for...of` iterates by Unicode code point (handling surrogate pairs),
 * matching Go runes and the TS code-point iteration; bias adaptation,
 * generalized base-36 digits, and the Number.MAX_SAFE_INTEGER overflow guard all
 * map 1:1. This is essentially the TS lib with the types erased, kept as the
 * dependency-free, no-build-step JavaScript twin.
 */

'use strict';

// RFC 3492 parameters (section 5), matching the TS/Go constants.
const BASE = 36;
const TMIN = 1;
const TMAX = 26;
const SKEW = 38;
const DAMP = 700;
const INITIAL_BIAS = 72;
const INITIAL_N = 128;
const ACE_PREFIX = 'xn--';
const MAX_INT = Number.MAX_SAFE_INTEGER; // 2^53-1 overflow guard for malformed input

/**
 * Bias adaptation (RFC 3492 section 6.1). delta/numpoints are always
 * non-negative, so Math.floor matches the reference integer division.
 */
function adapt(delta, numpoints, firsttime) {
  let d = firsttime ? Math.floor(delta / DAMP) : Math.floor(delta / 2);
  d += Math.floor(d / numpoints);
  let k = 0;
  while (d > Math.floor(((BASE - TMIN) * TMAX) / 2)) {
    d = Math.floor(d / (BASE - TMIN));
    k += BASE;
  }
  return k + Math.floor(((BASE - TMIN + 1) * d) / (d + SKEW));
}

/** Map a digit value (0-35) to its RFC 3492 base-36 character (lowercase). */
function digitToChar(d) {
  return d < 26
    ? String.fromCharCode(97 + d)            // a-z
    : String.fromCharCode(48 + (d - 26));    // 0-9
}

/** Map a character to its digit value (0-35), case-insensitive, or -1 if invalid. */
function charToDigit(c) {
  const code = c.charCodeAt(0);
  if (code >= 97 && code <= 122) return code - 97;      // a-z
  if (code >= 65 && code <= 90) return code - 65;       // A-Z
  if (code >= 48 && code <= 57) return code - 48 + 26;  // 0-9
  return -1;
}

/** True if the string contains any non-ASCII code point (>= 128). Surrogates
 *  (astral chars) are >= 128, so this matches the TS unit scan exactly. */
function hasNonAscii(s) {
  for (const ch of s) {
    if (ch.codePointAt(0) >= 128) return true;
  }
  return false;
}

/** Safely turn a decoded code point into a string. Mirrors Go's rune(n) (which
 *  emits U+FFFD for an out-of-range or surrogate value) rather than throwing. */
function fromCodePointSafe(n) {
  const valid = n >= 0 && n <= 0x10FFFF && !(n >= 0xD800 && n <= 0xDFFF);
  return valid ? String.fromCodePoint(n) : '�';
}

/**
 * Punycode-encode a single label (RFC 3492). Returns the encoded label with no
 * ACE prefix. Basic (ASCII) code points are emitted first, followed by a '-'
 * delimiter (only if there was at least one), then the generalized base-36
 * deltas for the non-basic code points. Twin of encodeLabel() in the TS/Go.
 */
function encodeLabel(input) {
  const codePoints = [];
  for (const ch of input) codePoints.push(ch.codePointAt(0)); // iterate by code point
  const length = codePoints.length;

  const output = [];
  for (const cp of codePoints) {
    if (cp < 128) output.push(String.fromCodePoint(cp));
  }
  const b = output.length;
  if (b > 0) output.push('-');

  let n = INITIAL_N;
  let delta = 0;
  let bias = INITIAL_BIAS;
  let h = b;

  while (h < length) {
    // Smallest code point in the input that is >= n.
    let m = Infinity;
    for (const cp of codePoints) if (cp >= n && cp < m) m = cp;
    delta += (m - n) * (h + 1);
    n = m;
    for (const cp of codePoints) {
      if (cp < n) {
        delta += 1;
      } else if (cp === n) {
        let q = delta;
        for (let k = BASE; ; k += BASE) {
          const t = Math.max(TMIN, Math.min(TMAX, k - bias));
          if (q < t) break;
          output.push(digitToChar(t + ((q - t) % (BASE - t))));
          q = Math.floor((q - t) / (BASE - t));
        }
        output.push(digitToChar(q));
        bias = adapt(delta, h + 1, h === b);
        delta = 0;
        h += 1;
      }
    }
    delta += 1;
    n += 1;
  }

  return output.join('');
}

/**
 * Punycode-decode a single label (RFC 3492). Returns the decoded label, or
 * `null` if the input is malformed (invalid digit, truncated generalized
 * number, non-ASCII in the basic portion, or arithmetic overflow). Twin of
 * decodeLabel() in the TS (string | null) and Go (bool form).
 */
function decodeLabel(input) {
  const lastDash = input.lastIndexOf('-');
  const output = [];
  if (lastDash >= 0) {
    for (let i = 0; i < lastDash; i++) {
      if (input.charCodeAt(i) >= 128) return null; // basic portion must be ASCII
      output.push(input[i]);
    }
  }
  const ext = lastDash >= 0 ? input.slice(lastDash + 1) : input;

  let n = INITIAL_N;
  let i = 0;
  let bias = INITIAL_BIAS;
  let pos = 0;

  while (pos < ext.length) {
    const oldi = i;
    let w = 1;
    for (let k = BASE; ; k += BASE) {
      if (pos >= ext.length) return null;        // truncated generalized number
      const digit = charToDigit(ext[pos]);
      if (digit < 0) return null;                // invalid digit
      pos += 1;
      if (digit >= MAX_INT / w) return null;     // overflow guard
      i += digit * w;
      const t = Math.max(TMIN, Math.min(TMAX, k - bias));
      if (digit < t) break;
      w *= BASE - t;
    }
    bias = adapt(i - oldi, output.length + 1, oldi === 0);
    const outLen = output.length + 1;
    n += Math.floor(i / outLen);
    i %= outLen;
    output.splice(i, 0, fromCodePointSafe(n));
    i += 1;
  }

  return output.join('');
}

/**
 * IDNA toASCII: encode a domain to Punycode ("xn--") form. Lowercases the whole
 * domain, splits on ".", ACE-encodes any label containing a non-ASCII code
 * point, leaves ASCII-only labels untouched, and rejoins with ".". Empty input
 * returns empty. Twin of encode() in the TS/Go.
 */
function encode(domain) {
  if (domain.length === 0) return '';
  const lower = domain.toLowerCase();
  return lower
    .split('.')
    .map((label) => (hasNonAscii(label) ? ACE_PREFIX + encodeLabel(label) : label))
    .join('.');
}

/**
 * IDNA toUnicode: decode a Punycode ("xn--") domain back to Unicode. Splits on
 * ".", decodes any label beginning with "xn--" (case-insensitive, detected on
 * the lowercased label), leaves every other label untouched, and rejoins with
 * ".". Returns `null` if any "xn--" label is invalid - the whole domain is
 * rejected, matching IDNA semantics. Empty input returns ''. Twin of decode()
 * in the TS/Go.
 */
function decode(domain) {
  if (domain.length === 0) return '';
  const out = [];
  for (const label of domain.split('.')) {
    if (label.toLowerCase().startsWith(ACE_PREFIX) && label.length > ACE_PREFIX.length) {
      const decoded = decodeLabel(label.slice(ACE_PREFIX.length));
      if (decoded === null) return null;
      out.push(decoded);
    } else {
      out.push(label);
    }
  }
  return out.join('.');
}

// CommonJS export so the file is consumable from Node without a build step,
// while staying dependency-free and framework-agnostic.
module.exports = { encode, decode, encodeLabel, decodeLabel };

// Showcase vectors - run only when executed directly via `node javascript.js`
// (not when require()'d as a library). Shared with the TS/Go/Rust/PHP/Python
// twins so every implementation is held to one contract.
if (require.main === module) {
  const assert = require('assert');
  assert.strictEqual(encode('münchen.de'), 'xn--mnchen-3ya.de');
  assert.strictEqual(decode('xn--mnchen-3ya.de'), 'münchen.de');
  assert.strictEqual(encode('Bücher.DE'), 'xn--bcher-kva.de'); // lowercased first
  assert.strictEqual(encodeLabel('café'), 'caf-dma');
  assert.strictEqual(decodeLabel('caf-dma'), 'café');
  assert.strictEqual(decode('xn--!'), null);             // invalid digit
  assert.strictEqual(decodeLabel('9'.repeat(40)), null); // overflow guard
  assert.strictEqual(decode(encode('café.fr')), 'café.fr');
  console.log('punycode: all showcase vectors passed');
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →