Skip to content

Punycode Converter — C++ source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//
// Language:   C++ (C++17, standard library only)
// Source:     CosmoDev polyglot showcase port of the Punycode tool, ported from
//             src/lib/punycode.ts (the canonical TypeScript implementation) and
//             cli/punycode/punycode.go (the live Go CLI twin).
// License:    display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; encode never fails, decode returns std::nullopt
//     on malformed input.
//   - Functionally equivalent to the TS/Go reference: same inputs -> same
//     outputs.
//   - Self-contained: stdlib only. std::u32string carries the code points so
//     astral characters (emoji, CJK extensions) are single elements, matching
//     Go runes and the TS code-point iteration. int64_t carries the RFC 3492
//     arithmetic; the 2^53-1 (Number.MAX_SAFE_INTEGER) overflow guard and code
//     points beyond U+10FFFF reject the label. Lowercasing is ASCII-only,
//     which covers IDNA host-label syntax.

#include <algorithm>
#include <cstdint>
#include <optional>
#include <string>
#include <vector>

namespace punycode {

constexpr int64_t kBase = 36;
constexpr int64_t kTmin = 1;
constexpr int64_t kTmax = 26;
constexpr int64_t kSkew = 38;
constexpr int64_t kDamp = 700;
constexpr int64_t kInitialBias = 72;
constexpr int64_t kInitialN = 128;
constexpr int64_t kMaxInt = 9007199254740991;  // 2^53-1 overflow guard
const std::string kAcePrefix = "xn--";

// Bias adaptation (RFC 3492 section 6.1).
int64_t adapt(int64_t delta, int64_t numpoints, bool firsttime) {
    int64_t d = firsttime ? delta / kDamp : delta / 2;
    d += d / numpoints;
    int64_t k = 0;
    while (d > (kBase - kTmin) * kTmax / 2) {
        d /= kBase - kTmin;
        k += kBase;
    }
    return k + (kBase - kTmin + 1) * d / (d + kSkew);
}

// Map a digit value (0-35) to its base-36 character (lowercase).
char digitToChar(int64_t d) { return d < 26 ? char('a' + d) : char('0' + (d - 26)); }

// Map a character to its digit value (0-35), case-insensitive, or -1 if invalid.
int64_t charToDigit(char c) {
    unsigned char u = static_cast<unsigned char>(c);
    if (u >= 'a' && u <= 'z') return u - 'a';
    if (u >= 'A' && u <= 'Z') return u - 'A';
    if (u >= '0' && u <= '9') return u - '0' + 26;
    return -1;
}

// Decode a UTF-8 string into code points (astral chars become one element).
std::u32string toCodePoints(const std::string& s) {
    std::u32string cps;
    for (size_t i = 0; i < s.size();) {
        unsigned char b = static_cast<unsigned char>(s[i]);
        if (b < 0x80) { cps.push_back(b); i += 1; }
        else if ((b >> 5) == 0x6 && i + 1 < s.size()) {  // 110xxxxx 10xxxxxx
            cps.push_back(((b & 0x1F) << 6) | (s[i + 1] & 0x3F));
            i += 2;
        } else if ((b >> 4) == 0xE && i + 2 < s.size()) {  // 3-byte
            cps.push_back(((b & 0x0F) << 12) | ((s[i + 1] & 0x3F) << 6) | (s[i + 2] & 0x3F));
            i += 3;
        } else if ((b >> 3) == 0x1E && i + 3 < s.size()) {  // 4-byte
            cps.push_back(((b & 0x07) << 18) | ((s[i + 1] & 0x3F) << 12) |
                          ((s[i + 2] & 0x3F) << 6) | (s[i + 3] & 0x3F));
            i += 4;
        } else {
            cps.push_back(b);  // lone byte: keep iteration terminating
            i += 1;
        }
    }
    return cps;
}

// Append a code point to a UTF-8 string.
void appendCodePoint(std::string& out, char32_t cp) {
    if (cp < 0x80) {
        out += char(cp);
    } else if (cp < 0x800) {
        out += char(0xC0 | (cp >> 6));
        out += char(0x80 | (cp & 0x3F));
    } else if (cp < 0x10000) {
        out += char(0xE0 | (cp >> 12));
        out += char(0x80 | ((cp >> 6) & 0x3F));
        out += char(0x80 | (cp & 0x3F));
    } else {
        out += char(0xF0 | (cp >> 18));
        out += char(0x80 | ((cp >> 12) & 0x3F));
        out += char(0x80 | ((cp >> 6) & 0x3F));
        out += char(0x80 | (cp & 0x3F));
    }
}

// True if the string contains any non-ASCII code point (>= 128).
bool hasNonAscii(const std::string& s) {
    for (char c : s)
        if (static_cast<unsigned char>(c) >= 128) return true;
    return false;
}

// Punycode-encode a single label (RFC 3492), no ACE prefix. Basic code points
// are emitted first, then a `-` delimiter (only if there was at least one),
// then the generalized-base-36 deltas.
std::string encodeLabel(const std::string& input) {
    const std::u32string cps = toCodePoints(input);
    const int64_t length = static_cast<int64_t>(cps.size());

    std::string out;
    int64_t b = 0;
    for (char32_t cp : cps)
        if (cp < 128) {
            out += char(cp);
            b++;
        }
    if (b > 0) out += '-';

    int64_t n = kInitialN, delta = 0, bias = kInitialBias, h = b;
    while (h < length) {
        int64_t m = INT64_MAX;
        for (char32_t cp : cps)
            if (cp >= n && cp < m) m = cp;  // smallest code point >= n
        delta += (m - n) * (h + 1);
        n = m;
        for (char32_t cp : cps) {
            if (cp < n) {
                delta += 1;
            } else if (cp == n) {
                int64_t q = delta;
                for (int64_t k = kBase;; k += kBase) {
                    int64_t t = std::max(kTmin, std::min(kTmax, k - bias));
                    if (q < t) break;
                    out += digitToChar(t + (q - t) % (kBase - t));
                    q = (q - t) / (kBase - t);
                }
                out += digitToChar(q);
                bias = adapt(delta, h + 1, h == b);
                delta = 0;
                h += 1;
            }
        }
        delta += 1;
        n += 1;
    }
    return out;
}

// Punycode-decode a single label (RFC 3492). Returns std::nullopt if the
// input is malformed (invalid digit, truncated generalized number, non-ASCII
// in the basic portion, code point beyond U+10FFFF, or overflow).
std::optional<std::string> decodeLabel(const std::string& input) {
    size_t lastDash = input.rfind('-');
    std::u32string out;
    if (lastDash != std::string::npos) {
        for (size_t i = 0; i < lastDash; i++) {
            if (static_cast<unsigned char>(input[i]) >= 128) return std::nullopt;  // basic portion must be ASCII
            out.push_back(input[i]);
        }
    }
    const std::string ext = lastDash != std::string::npos ? input.substr(lastDash + 1) : input;

    int64_t n = kInitialN, i = 0, bias = kInitialBias;
    size_t pos = 0;
    while (pos < ext.size()) {
        int64_t oldi = i, w = 1;
        for (int64_t k = kBase;; k += kBase) {
            if (pos >= ext.size()) return std::nullopt;  // truncated generalized number
            int64_t digit = charToDigit(ext[pos]);
            if (digit < 0) return std::nullopt;  // invalid digit
            pos += 1;
            if (digit >= kMaxInt / w) return std::nullopt;  // overflow guard
            i += digit * w;
            int64_t t = std::max(kTmin, std::min(kTmax, k - bias));
            if (digit < t) break;
            w *= kBase - t;
        }
        bias = adapt(i - oldi, static_cast<int64_t>(out.size()) + 1, oldi == 0);
        int64_t outLen = static_cast<int64_t>(out.size()) + 1;
        n += i / outLen;
        i %= outLen;
        if (n > 0x10FFFF) return std::nullopt;
        out.insert(out.begin() + i, static_cast<char32_t>(n));
        i += 1;
    }

    std::string result;
    for (char32_t cp : out) appendCodePoint(result, cp);
    return result;
}

// Split on '.', keeping empty labels (including a trailing one), like the TS
// String.split('.').
std::vector<std::string> splitLabels(const std::string& s) {
    std::vector<std::string> labels;
    std::string cur;
    for (char c : s) {
        if (c == '.') {
            labels.push_back(cur);
            cur.clear();
        } else {
            cur += c;
        }
    }
    labels.push_back(cur);
    return labels;
}

// IDNA toASCII: lowercase the domain, ACE-encode ("xn--" + Punycode) any label
// containing a non-ASCII code point, leave ASCII-only labels untouched.
// Empty input returns empty.
std::string encode(const std::string& domain) {
    if (domain.empty()) return "";
    std::string lower = domain;
    for (char& c : lower)
        if (c >= 'A' && c <= 'Z') c = char(c - 'A' + 'a');
    std::string out;
    bool first = true;
    for (const std::string& label : splitLabels(lower)) {
        if (!first) out += '.';
        first = false;
        out += hasNonAscii(label) ? kAcePrefix + encodeLabel(label) : label;
    }
    return out;
}

// IDNA toUnicode: decode any "xn--" label (case-insensitive, prefix detected
// on the lowercased label), leave every other label untouched. Returns
// std::nullopt when any "xn--" label is invalid - the whole domain is
// rejected, matching IDNA semantics. Empty input returns empty.
std::optional<std::string> decode(const std::string& domain) {
    if (domain.empty()) return "";
    std::string out;
    bool first = true;
    for (const std::string& label : splitLabels(domain)) {
        if (!first) out += '.';
        first = false;
        std::string low = label;
        for (char& c : low)
            if (c >= 'A' && c <= 'Z') c = char(c - 'A' + 'a');
        if (low.rfind(kAcePrefix, 0) == 0 && label.size() > kAcePrefix.size()) {
            std::optional<std::string> decoded = decodeLabel(label.substr(kAcePrefix.size()));
            if (!decoded) return std::nullopt;
            out += *decoded;
        } else {
            out += label;
        }
    }
    return out;
}

}  // namespace punycode

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →