Punycode Converter — C source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/*
* punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
*
* Language: C (C99, standard library only)
* Source: CosmoDev polyglot showcase port of the Punycode tool, ported from
* src/lib/punycode.ts (the canonical TypeScript implementation) and
* cli/punycode/punycode.go (the live Go CLI twin).
* License: display source - part of CosmoDev's polyglot tool pages.
*
* Design goals:
* - Pure + deterministic; encode never fails, decode returns NULL on
* malformed input (never crashes).
* - Functionally equivalent to the TS/Go reference: same inputs -> same
* outputs.
* - Self-contained: stdlib only. Strings are UTF-8; code points are
* iterated/encoded with two small helpers so astral characters (emoji,
* CJK extensions) are single elements, matching Go runes and the TS
* code-point iteration. The 2^53-1 (Number.MAX_SAFE_INTEGER) overflow
* guard and code points beyond U+10FFFF reject the label. Lowercasing
* is ASCII-only, which covers IDNA host-label syntax.
*/
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#define PC_BASE 36
#define PC_TMIN 1
#define PC_TMAX 26
#define PC_SKEW 38
#define PC_DAMP 700
#define PC_INITIAL_BIAS 72
#define PC_INITIAL_N 128
#define PC_ACE_PREFIX "xn--"
#define PC_MAX_INT ((int64_t)9007199254740991) /* 2^53-1 overflow guard */
/* Growable UTF-8 output buffer; pc_str_free() owns what it returns. */
typedef struct {
char *p;
size_t len, cap;
} PcStr;
static bool pc_str_push(PcStr *s, const char *bytes, size_t n) {
if (s->len + n > s->cap) {
size_t cap = s->cap ? s->cap * 2 : 64;
while (cap < s->len + n) cap *= 2;
char *q = realloc(s->p, cap);
if (!q) return false;
s->p = q, s->cap = cap;
}
memcpy(s->p + s->len, bytes, n);
s->len += n;
return true;
}
/* Decode the next UTF-8 code point at *pp (advancing it); a lone invalid
* byte is taken as its own value so iteration always terminates. */
static uint32_t pc_utf8_next(const char **pp) {
const unsigned char *s = (const unsigned char *)*pp;
uint32_t c = s[0];
int n = c >= 0xF0 ? 4 : c >= 0xE0 ? 3 : c >= 0xC0 ? 2 : 1;
uint32_t cp = c >= 0xF0 ? c & 0x07 : c >= 0xE0 ? c & 0x0F : c >= 0xC0 ? c & 0x1F : c;
for (int i = 1; i < n && (s[i] & 0xC0) == 0x80; i++) cp = (cp << 6) | (s[i] & 0x3F);
*pp += n;
return cp;
}
/* Append code point cp as UTF-8; false when cp is beyond U+10FFFF. */
static bool pc_utf8_append(PcStr *s, uint32_t cp) {
char b[4];
int n;
if (cp < 0x80) { b[0] = (char)cp; n = 1; }
else if (cp < 0x800) { b[0] = (char)(0xC0 | cp >> 6); b[1] = (char)(0x80 | (cp & 0x3F)); n = 2; }
else if (cp < 0x10000) { b[0] = (char)(0xE0 | cp >> 12); b[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); b[2] = (char)(0x80 | (cp & 0x3F)); n = 3; }
else if (cp <= 0x10FFFF) { b[0] = (char)(0xF0 | cp >> 18); b[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); b[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); b[3] = (char)(0x80 | (cp & 0x3F)); n = 4; }
else return false;
return pc_str_push(s, b, (size_t)n);
}
/* Bias adaptation (RFC 3492 section 6.1). */
static int64_t pc_adapt(int64_t delta, int64_t numpoints, bool firsttime) {
int64_t d = firsttime ? delta / PC_DAMP : delta / 2;
d += d / numpoints;
int64_t k = 0;
while (d > (int64_t)(PC_BASE - PC_TMIN) * PC_TMAX / 2) {
d /= PC_BASE - PC_TMIN;
k += PC_BASE;
}
return k + (int64_t)(PC_BASE - PC_TMIN + 1) * d / (d + PC_SKEW);
}
/* Map a byte to its digit value (0-35), case-insensitive, or -1 if invalid. */
static int64_t pc_char_to_digit(unsigned char c) {
if (c >= 'a' && c <= 'z') return c - 'a';
if (c >= 'A' && c <= 'Z') return c - 'A';
if (c >= '0' && c <= '9') return c - '0' + 26;
return -1;
}
static char pc_digit_to_char(int64_t d) {
return d < 26 ? (char)('a' + d) : (char)('0' + (d - 26));
}
/* Punycode-encode a single label (RFC 3492), no ACE prefix. Fails only on
* allocation. Basic code points are emitted first, then a `-` delimiter
* (only if there was at least one), then the base-36 deltas. */
static bool pc_encode_label(const char *input, PcStr *out) {
uint32_t cps[512];
size_t length = 0;
for (const char *p = input; *p; ) {
if (length >= 512) return false; /* beyond any realistic label */
cps[length++] = pc_utf8_next(&p);
}
size_t b = 0;
for (size_t i = 0; i < length; i++)
if (cps[i] < 128) {
char c = (char)cps[i];
if (!pc_str_push(out, &c, 1)) return false;
b++;
}
if (b > 0) {
char dash = '-';
if (!pc_str_push(out, &dash, 1)) return false;
}
int64_t n = PC_INITIAL_N, delta = 0, bias = PC_INITIAL_BIAS, h = (int64_t)b;
while (h < (int64_t)length) {
int64_t m = INT64_MAX;
for (size_t i = 0; i < length; i++)
if (cps[i] >= (uint32_t)n && cps[i] < (uint32_t)m) m = cps[i];
delta += (m - n) * (h + 1);
n = m;
for (size_t i = 0; i < length; i++) {
if (cps[i] < (uint32_t)n) {
delta += 1;
} else if (cps[i] == (uint32_t)n) {
int64_t q = delta;
for (int64_t k = PC_BASE; ; k += PC_BASE) {
int64_t t = k - bias;
if (t < PC_TMIN) t = PC_TMIN;
if (t > PC_TMAX) t = PC_TMAX;
if (q < t) break;
char c = pc_digit_to_char(t + (q - t) % (PC_BASE - t));
if (!pc_str_push(out, &c, 1)) return false;
q = (q - t) / (PC_BASE - t);
}
char c = pc_digit_to_char(q);
if (!pc_str_push(out, &c, 1)) return false;
bias = pc_adapt(delta, h + 1, h == (int64_t)b);
delta = 0;
h += 1;
}
}
delta += 1;
n += 1;
}
return true;
}
/* Punycode-decode a single label (RFC 3492). Returns false on malformed
* input (invalid digit, truncated generalized number, non-ASCII in the
* basic portion, code point beyond U+10FFFF, overflow) or allocation. */
static bool pc_decode_label(const char *input, PcStr *out) {
size_t in_len = strlen(input);
const char *dash = input ? strrchr(input, '-') : NULL; /* last '-', may be basic/ext divider */
size_t last_dash = dash ? (size_t)(dash - input) : (size_t)-1;
uint32_t *cps = malloc((in_len + 1) * sizeof *cps); /* output cps (never more than input digits + basic) */
if (!cps) return false;
size_t out_len = 0;
if (last_dash != (size_t)-1) {
for (size_t i = 0; i < last_dash; i++) {
if ((unsigned char)input[i] >= 128) { free(cps); return false; } /* basic portion must be ASCII */
cps[out_len++] = (unsigned char)input[i];
}
}
const char *ext = last_dash != (size_t)-1 ? input + last_dash + 1 : input;
int64_t n = PC_INITIAL_N, i = 0, bias = PC_INITIAL_BIAS;
const char *pos = ext;
while (*pos) {
int64_t oldi = i, w = 1;
for (int64_t k = PC_BASE; ; k += PC_BASE) {
if (!*pos) { free(cps); return false; } /* truncated generalized number */
int64_t digit = pc_char_to_digit((unsigned char)*pos++);
if (digit < 0) { free(cps); return false; }
if (digit >= PC_MAX_INT / w) { free(cps); return false; } /* overflow guard */
i += digit * w;
int64_t t = k - bias;
if (t < PC_TMIN) t = PC_TMIN;
if (t > PC_TMAX) t = PC_TMAX;
if (digit < t) break;
w *= PC_BASE - t;
}
bias = pc_adapt(i - oldi, (int64_t)out_len + 1, oldi == 0);
int64_t out_len_1 = (int64_t)out_len + 1;
n += i / out_len_1;
i %= out_len_1;
if (n > 0x10FFFF) { free(cps); return false; }
memmove(cps + i + 1, cps + i, (out_len - (size_t)i) * sizeof *cps); /* splice at i */
cps[i] = (uint32_t)n;
out_len++;
i += 1;
}
bool ok = true;
for (size_t j = 0; j < out_len && ok; j++) ok = pc_utf8_append(out, cps[j]);
free(cps);
return ok;
}
/* True if the UTF-8 string contains any non-ASCII code point. */
static bool pc_has_non_ascii(const char *s) {
for (const unsigned char *p = (const unsigned char *)s; *p; p++)
if (*p >= 128) return true;
return false;
}
/* IDNA toASCII: lowercase, split on '.', ACE-encode ("xn--" + Punycode) any
* label containing a non-ASCII code point, rejoin (empty labels, including a
* trailing one, pass through as in the TS split('.')). Returns a malloc'd
* NUL-terminated string (the caller frees) or NULL on allocation failure.
* Empty input -> "". */
char *punycode_encode(const char *domain) {
if (!*domain) return strdup("");
PcStr out = {0};
char *lower = strdup(domain);
if (!lower) return NULL;
for (char *p = lower; *p; p++)
if (*p >= 'A' && *p <= 'Z') *p += 'a' - 'A';
bool ok = true, first = true;
const char *p = lower;
for (;;) {
const char *dot = strchr(p, '.');
size_t ll = dot ? (size_t)(dot - p) : strlen(p);
if (ok && !first) {
char d = '.';
ok = pc_str_push(&out, &d, 1);
}
char *lab = strndup(p, ll); /* bound the label: helpers read to NUL */
if (!lab) { ok = false; break; }
if (pc_has_non_ascii(lab)) {
PcStr enc = {0};
ok = ok && pc_str_push(&out, PC_ACE_PREFIX, 4) &&
pc_encode_label(lab, &enc) && pc_str_push(&out, enc.p, enc.len);
free(enc.p);
} else {
ok = ok && pc_str_push(&out, lab, ll);
}
free(lab);
if (!dot || !ok) break;
first = false;
p = dot + 1;
}
free(lower);
if (!ok) { free(out.p); return NULL; }
if (!pc_str_push(&out, "", 1)) { free(out.p); return NULL; } /* NUL */
return out.p;
}
/* IDNA toUnicode: split on '.', decode any "xn--" label (case-insensitive,
* prefix detected on the lowercased label), reject the whole domain when one
* is invalid. Returns a malloc'd NUL-terminated string or NULL (malformed /
* allocation). Empty input -> "". */
char *punycode_decode(const char *domain) {
if (!*domain) return strdup("");
PcStr out = {0};
bool ok = true, first = true;
const char *p = domain;
for (;;) {
const char *dot = strchr(p, '.');
size_t ll = dot ? (size_t)(dot - p) : strlen(p);
if (ok && !first) {
char d = '.';
ok = pc_str_push(&out, &d, 1);
}
char low[64]; /* lowercased prefix of the label, for the ACE check */
size_t ml = ll < 63 ? ll : 63;
for (size_t i = 0; i < ml; i++) {
char c = p[i];
low[i] = (c >= 'A' && c <= 'Z') ? (char)(c + 'a' - 'A') : c;
}
if (ll > 4 && strncmp(low, PC_ACE_PREFIX, 4) == 0) {
char *lab = strndup(p + 4, ll - 4);
PcStr dec = {0};
ok = ok && lab && pc_decode_label(lab, &dec) &&
pc_str_push(&out, dec.p, dec.len);
free(lab);
free(dec.p);
} else {
ok = ok && pc_str_push(&out, p, ll);
}
if (!dot || !ok) break;
first = false;
p = dot + 1;
}
if (!ok) { free(out.p); return NULL; }
if (!pc_str_push(&out, "", 1)) { free(out.p); return NULL; } /* NUL */
return out.p;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →