Punycode Converter — C++ source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//
// Language: C++ (C++17, standard library only)
// Source: CosmoDev polyglot showcase port of the Punycode tool, ported from
// src/lib/punycode.ts (the canonical TypeScript implementation) and
// cli/punycode/punycode.go (the live Go CLI twin).
// License: display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; encode never fails, decode returns std::nullopt
// on malformed input.
// - Functionally equivalent to the TS/Go reference: same inputs -> same
// outputs.
// - Self-contained: stdlib only. std::u32string carries the code points so
// astral characters (emoji, CJK extensions) are single elements, matching
// Go runes and the TS code-point iteration. int64_t carries the RFC 3492
// arithmetic; the 2^53-1 (Number.MAX_SAFE_INTEGER) overflow guard and code
// points beyond U+10FFFF reject the label. Lowercasing is ASCII-only,
// which covers IDNA host-label syntax.
#include <algorithm>
#include <cstdint>
#include <optional>
#include <string>
#include <vector>
namespace punycode {
constexpr int64_t kBase = 36;
constexpr int64_t kTmin = 1;
constexpr int64_t kTmax = 26;
constexpr int64_t kSkew = 38;
constexpr int64_t kDamp = 700;
constexpr int64_t kInitialBias = 72;
constexpr int64_t kInitialN = 128;
constexpr int64_t kMaxInt = 9007199254740991; // 2^53-1 overflow guard
const std::string kAcePrefix = "xn--";
// Bias adaptation (RFC 3492 section 6.1).
int64_t adapt(int64_t delta, int64_t numpoints, bool firsttime) {
int64_t d = firsttime ? delta / kDamp : delta / 2;
d += d / numpoints;
int64_t k = 0;
while (d > (kBase - kTmin) * kTmax / 2) {
d /= kBase - kTmin;
k += kBase;
}
return k + (kBase - kTmin + 1) * d / (d + kSkew);
}
// Map a digit value (0-35) to its base-36 character (lowercase).
char digitToChar(int64_t d) { return d < 26 ? char('a' + d) : char('0' + (d - 26)); }
// Map a character to its digit value (0-35), case-insensitive, or -1 if invalid.
int64_t charToDigit(char c) {
unsigned char u = static_cast<unsigned char>(c);
if (u >= 'a' && u <= 'z') return u - 'a';
if (u >= 'A' && u <= 'Z') return u - 'A';
if (u >= '0' && u <= '9') return u - '0' + 26;
return -1;
}
// Decode a UTF-8 string into code points (astral chars become one element).
std::u32string toCodePoints(const std::string& s) {
std::u32string cps;
for (size_t i = 0; i < s.size();) {
unsigned char b = static_cast<unsigned char>(s[i]);
if (b < 0x80) { cps.push_back(b); i += 1; }
else if ((b >> 5) == 0x6 && i + 1 < s.size()) { // 110xxxxx 10xxxxxx
cps.push_back(((b & 0x1F) << 6) | (s[i + 1] & 0x3F));
i += 2;
} else if ((b >> 4) == 0xE && i + 2 < s.size()) { // 3-byte
cps.push_back(((b & 0x0F) << 12) | ((s[i + 1] & 0x3F) << 6) | (s[i + 2] & 0x3F));
i += 3;
} else if ((b >> 3) == 0x1E && i + 3 < s.size()) { // 4-byte
cps.push_back(((b & 0x07) << 18) | ((s[i + 1] & 0x3F) << 12) |
((s[i + 2] & 0x3F) << 6) | (s[i + 3] & 0x3F));
i += 4;
} else {
cps.push_back(b); // lone byte: keep iteration terminating
i += 1;
}
}
return cps;
}
// Append a code point to a UTF-8 string.
void appendCodePoint(std::string& out, char32_t cp) {
if (cp < 0x80) {
out += char(cp);
} else if (cp < 0x800) {
out += char(0xC0 | (cp >> 6));
out += char(0x80 | (cp & 0x3F));
} else if (cp < 0x10000) {
out += char(0xE0 | (cp >> 12));
out += char(0x80 | ((cp >> 6) & 0x3F));
out += char(0x80 | (cp & 0x3F));
} else {
out += char(0xF0 | (cp >> 18));
out += char(0x80 | ((cp >> 12) & 0x3F));
out += char(0x80 | ((cp >> 6) & 0x3F));
out += char(0x80 | (cp & 0x3F));
}
}
// True if the string contains any non-ASCII code point (>= 128).
bool hasNonAscii(const std::string& s) {
for (char c : s)
if (static_cast<unsigned char>(c) >= 128) return true;
return false;
}
// Punycode-encode a single label (RFC 3492), no ACE prefix. Basic code points
// are emitted first, then a `-` delimiter (only if there was at least one),
// then the generalized-base-36 deltas.
std::string encodeLabel(const std::string& input) {
const std::u32string cps = toCodePoints(input);
const int64_t length = static_cast<int64_t>(cps.size());
std::string out;
int64_t b = 0;
for (char32_t cp : cps)
if (cp < 128) {
out += char(cp);
b++;
}
if (b > 0) out += '-';
int64_t n = kInitialN, delta = 0, bias = kInitialBias, h = b;
while (h < length) {
int64_t m = INT64_MAX;
for (char32_t cp : cps)
if (cp >= n && cp < m) m = cp; // smallest code point >= n
delta += (m - n) * (h + 1);
n = m;
for (char32_t cp : cps) {
if (cp < n) {
delta += 1;
} else if (cp == n) {
int64_t q = delta;
for (int64_t k = kBase;; k += kBase) {
int64_t t = std::max(kTmin, std::min(kTmax, k - bias));
if (q < t) break;
out += digitToChar(t + (q - t) % (kBase - t));
q = (q - t) / (kBase - t);
}
out += digitToChar(q);
bias = adapt(delta, h + 1, h == b);
delta = 0;
h += 1;
}
}
delta += 1;
n += 1;
}
return out;
}
// Punycode-decode a single label (RFC 3492). Returns std::nullopt if the
// input is malformed (invalid digit, truncated generalized number, non-ASCII
// in the basic portion, code point beyond U+10FFFF, or overflow).
std::optional<std::string> decodeLabel(const std::string& input) {
size_t lastDash = input.rfind('-');
std::u32string out;
if (lastDash != std::string::npos) {
for (size_t i = 0; i < lastDash; i++) {
if (static_cast<unsigned char>(input[i]) >= 128) return std::nullopt; // basic portion must be ASCII
out.push_back(input[i]);
}
}
const std::string ext = lastDash != std::string::npos ? input.substr(lastDash + 1) : input;
int64_t n = kInitialN, i = 0, bias = kInitialBias;
size_t pos = 0;
while (pos < ext.size()) {
int64_t oldi = i, w = 1;
for (int64_t k = kBase;; k += kBase) {
if (pos >= ext.size()) return std::nullopt; // truncated generalized number
int64_t digit = charToDigit(ext[pos]);
if (digit < 0) return std::nullopt; // invalid digit
pos += 1;
if (digit >= kMaxInt / w) return std::nullopt; // overflow guard
i += digit * w;
int64_t t = std::max(kTmin, std::min(kTmax, k - bias));
if (digit < t) break;
w *= kBase - t;
}
bias = adapt(i - oldi, static_cast<int64_t>(out.size()) + 1, oldi == 0);
int64_t outLen = static_cast<int64_t>(out.size()) + 1;
n += i / outLen;
i %= outLen;
if (n > 0x10FFFF) return std::nullopt;
out.insert(out.begin() + i, static_cast<char32_t>(n));
i += 1;
}
std::string result;
for (char32_t cp : out) appendCodePoint(result, cp);
return result;
}
// Split on '.', keeping empty labels (including a trailing one), like the TS
// String.split('.').
std::vector<std::string> splitLabels(const std::string& s) {
std::vector<std::string> labels;
std::string cur;
for (char c : s) {
if (c == '.') {
labels.push_back(cur);
cur.clear();
} else {
cur += c;
}
}
labels.push_back(cur);
return labels;
}
// IDNA toASCII: lowercase the domain, ACE-encode ("xn--" + Punycode) any label
// containing a non-ASCII code point, leave ASCII-only labels untouched.
// Empty input returns empty.
std::string encode(const std::string& domain) {
if (domain.empty()) return "";
std::string lower = domain;
for (char& c : lower)
if (c >= 'A' && c <= 'Z') c = char(c - 'A' + 'a');
std::string out;
bool first = true;
for (const std::string& label : splitLabels(lower)) {
if (!first) out += '.';
first = false;
out += hasNonAscii(label) ? kAcePrefix + encodeLabel(label) : label;
}
return out;
}
// IDNA toUnicode: decode any "xn--" label (case-insensitive, prefix detected
// on the lowercased label), leave every other label untouched. Returns
// std::nullopt when any "xn--" label is invalid - the whole domain is
// rejected, matching IDNA semantics. Empty input returns empty.
std::optional<std::string> decode(const std::string& domain) {
if (domain.empty()) return "";
std::string out;
bool first = true;
for (const std::string& label : splitLabels(domain)) {
if (!first) out += '.';
first = false;
std::string low = label;
for (char& c : low)
if (c >= 'A' && c <= 'Z') c = char(c - 'A' + 'a');
if (low.rfind(kAcePrefix, 0) == 0 && label.size() > kAcePrefix.size()) {
std::optional<std::string> decoded = decodeLabel(label.substr(kAcePrefix.size()));
if (!decoded) return std::nullopt;
out += *decoded;
} else {
out += label;
}
}
return out;
}
} // namespace punycode
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →