Skip to content

Punycode Converter — C source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
 *
 * Language:   C (C99, standard library only)
 * Source:     CosmoDev polyglot showcase port of the Punycode tool, ported from
 *             src/lib/punycode.ts (the canonical TypeScript implementation) and
 *             cli/punycode/punycode.go (the live Go CLI twin).
 * License:    display source - part of CosmoDev's polyglot tool pages.
 *
 * Design goals:
 *   - Pure + deterministic; encode never fails, decode returns NULL on
 *     malformed input (never crashes).
 *   - Functionally equivalent to the TS/Go reference: same inputs -> same
 *     outputs.
 *   - Self-contained: stdlib only. Strings are UTF-8; code points are
 *     iterated/encoded with two small helpers so astral characters (emoji,
 *     CJK extensions) are single elements, matching Go runes and the TS
 *     code-point iteration. The 2^53-1 (Number.MAX_SAFE_INTEGER) overflow
 *     guard and code points beyond U+10FFFF reject the label. Lowercasing
 *     is ASCII-only, which covers IDNA host-label syntax.
 */

#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>

#define PC_BASE 36
#define PC_TMIN 1
#define PC_TMAX 26
#define PC_SKEW 38
#define PC_DAMP 700
#define PC_INITIAL_BIAS 72
#define PC_INITIAL_N 128
#define PC_ACE_PREFIX "xn--"
#define PC_MAX_INT ((int64_t)9007199254740991) /* 2^53-1 overflow guard */

/* Growable UTF-8 output buffer; pc_str_free() owns what it returns. */
typedef struct {
    char *p;
    size_t len, cap;
} PcStr;

static bool pc_str_push(PcStr *s, const char *bytes, size_t n) {
    if (s->len + n > s->cap) {
        size_t cap = s->cap ? s->cap * 2 : 64;
        while (cap < s->len + n) cap *= 2;
        char *q = realloc(s->p, cap);
        if (!q) return false;
        s->p = q, s->cap = cap;
    }
    memcpy(s->p + s->len, bytes, n);
    s->len += n;
    return true;
}

/* Decode the next UTF-8 code point at *pp (advancing it); a lone invalid
 * byte is taken as its own value so iteration always terminates. */
static uint32_t pc_utf8_next(const char **pp) {
    const unsigned char *s = (const unsigned char *)*pp;
    uint32_t c = s[0];
    int n = c >= 0xF0 ? 4 : c >= 0xE0 ? 3 : c >= 0xC0 ? 2 : 1;
    uint32_t cp = c >= 0xF0 ? c & 0x07 : c >= 0xE0 ? c & 0x0F : c >= 0xC0 ? c & 0x1F : c;
    for (int i = 1; i < n && (s[i] & 0xC0) == 0x80; i++) cp = (cp << 6) | (s[i] & 0x3F);
    *pp += n;
    return cp;
}

/* Append code point cp as UTF-8; false when cp is beyond U+10FFFF. */
static bool pc_utf8_append(PcStr *s, uint32_t cp) {
    char b[4];
    int n;
    if (cp < 0x80) { b[0] = (char)cp; n = 1; }
    else if (cp < 0x800) { b[0] = (char)(0xC0 | cp >> 6); b[1] = (char)(0x80 | (cp & 0x3F)); n = 2; }
    else if (cp < 0x10000) { b[0] = (char)(0xE0 | cp >> 12); b[1] = (char)(0x80 | ((cp >> 6) & 0x3F)); b[2] = (char)(0x80 | (cp & 0x3F)); n = 3; }
    else if (cp <= 0x10FFFF) { b[0] = (char)(0xF0 | cp >> 18); b[1] = (char)(0x80 | ((cp >> 12) & 0x3F)); b[2] = (char)(0x80 | ((cp >> 6) & 0x3F)); b[3] = (char)(0x80 | (cp & 0x3F)); n = 4; }
    else return false;
    return pc_str_push(s, b, (size_t)n);
}

/* Bias adaptation (RFC 3492 section 6.1). */
static int64_t pc_adapt(int64_t delta, int64_t numpoints, bool firsttime) {
    int64_t d = firsttime ? delta / PC_DAMP : delta / 2;
    d += d / numpoints;
    int64_t k = 0;
    while (d > (int64_t)(PC_BASE - PC_TMIN) * PC_TMAX / 2) {
        d /= PC_BASE - PC_TMIN;
        k += PC_BASE;
    }
    return k + (int64_t)(PC_BASE - PC_TMIN + 1) * d / (d + PC_SKEW);
}

/* Map a byte to its digit value (0-35), case-insensitive, or -1 if invalid. */
static int64_t pc_char_to_digit(unsigned char c) {
    if (c >= 'a' && c <= 'z') return c - 'a';
    if (c >= 'A' && c <= 'Z') return c - 'A';
    if (c >= '0' && c <= '9') return c - '0' + 26;
    return -1;
}

static char pc_digit_to_char(int64_t d) {
    return d < 26 ? (char)('a' + d) : (char)('0' + (d - 26));
}

/* Punycode-encode a single label (RFC 3492), no ACE prefix. Fails only on
 * allocation. Basic code points are emitted first, then a `-` delimiter
 * (only if there was at least one), then the base-36 deltas. */
static bool pc_encode_label(const char *input, PcStr *out) {
    uint32_t cps[512];
    size_t length = 0;
    for (const char *p = input; *p; ) {
        if (length >= 512) return false; /* beyond any realistic label */
        cps[length++] = pc_utf8_next(&p);
    }

    size_t b = 0;
    for (size_t i = 0; i < length; i++)
        if (cps[i] < 128) {
            char c = (char)cps[i];
            if (!pc_str_push(out, &c, 1)) return false;
            b++;
        }
    if (b > 0) {
        char dash = '-';
        if (!pc_str_push(out, &dash, 1)) return false;
    }

    int64_t n = PC_INITIAL_N, delta = 0, bias = PC_INITIAL_BIAS, h = (int64_t)b;
    while (h < (int64_t)length) {
        int64_t m = INT64_MAX;
        for (size_t i = 0; i < length; i++)
            if (cps[i] >= (uint32_t)n && cps[i] < (uint32_t)m) m = cps[i];
        delta += (m - n) * (h + 1);
        n = m;
        for (size_t i = 0; i < length; i++) {
            if (cps[i] < (uint32_t)n) {
                delta += 1;
            } else if (cps[i] == (uint32_t)n) {
                int64_t q = delta;
                for (int64_t k = PC_BASE; ; k += PC_BASE) {
                    int64_t t = k - bias;
                    if (t < PC_TMIN) t = PC_TMIN;
                    if (t > PC_TMAX) t = PC_TMAX;
                    if (q < t) break;
                    char c = pc_digit_to_char(t + (q - t) % (PC_BASE - t));
                    if (!pc_str_push(out, &c, 1)) return false;
                    q = (q - t) / (PC_BASE - t);
                }
                char c = pc_digit_to_char(q);
                if (!pc_str_push(out, &c, 1)) return false;
                bias = pc_adapt(delta, h + 1, h == (int64_t)b);
                delta = 0;
                h += 1;
            }
        }
        delta += 1;
        n += 1;
    }
    return true;
}

/* Punycode-decode a single label (RFC 3492). Returns false on malformed
 * input (invalid digit, truncated generalized number, non-ASCII in the
 * basic portion, code point beyond U+10FFFF, overflow) or allocation. */
static bool pc_decode_label(const char *input, PcStr *out) {
    size_t in_len = strlen(input);
    const char *dash = input ? strrchr(input, '-') : NULL; /* last '-', may be basic/ext divider */
    size_t last_dash = dash ? (size_t)(dash - input) : (size_t)-1;

    uint32_t *cps = malloc((in_len + 1) * sizeof *cps); /* output cps (never more than input digits + basic) */
    if (!cps) return false;
    size_t out_len = 0;

    if (last_dash != (size_t)-1) {
        for (size_t i = 0; i < last_dash; i++) {
            if ((unsigned char)input[i] >= 128) { free(cps); return false; } /* basic portion must be ASCII */
            cps[out_len++] = (unsigned char)input[i];
        }
    }
    const char *ext = last_dash != (size_t)-1 ? input + last_dash + 1 : input;

    int64_t n = PC_INITIAL_N, i = 0, bias = PC_INITIAL_BIAS;
    const char *pos = ext;
    while (*pos) {
        int64_t oldi = i, w = 1;
        for (int64_t k = PC_BASE; ; k += PC_BASE) {
            if (!*pos) { free(cps); return false; } /* truncated generalized number */
            int64_t digit = pc_char_to_digit((unsigned char)*pos++);
            if (digit < 0) { free(cps); return false; }
            if (digit >= PC_MAX_INT / w) { free(cps); return false; } /* overflow guard */
            i += digit * w;
            int64_t t = k - bias;
            if (t < PC_TMIN) t = PC_TMIN;
            if (t > PC_TMAX) t = PC_TMAX;
            if (digit < t) break;
            w *= PC_BASE - t;
        }
        bias = pc_adapt(i - oldi, (int64_t)out_len + 1, oldi == 0);
        int64_t out_len_1 = (int64_t)out_len + 1;
        n += i / out_len_1;
        i %= out_len_1;
        if (n > 0x10FFFF) { free(cps); return false; }
        memmove(cps + i + 1, cps + i, (out_len - (size_t)i) * sizeof *cps); /* splice at i */
        cps[i] = (uint32_t)n;
        out_len++;
        i += 1;
    }

    bool ok = true;
    for (size_t j = 0; j < out_len && ok; j++) ok = pc_utf8_append(out, cps[j]);
    free(cps);
    return ok;
}

/* True if the UTF-8 string contains any non-ASCII code point. */
static bool pc_has_non_ascii(const char *s) {
    for (const unsigned char *p = (const unsigned char *)s; *p; p++)
        if (*p >= 128) return true;
    return false;
}

/* IDNA toASCII: lowercase, split on '.', ACE-encode ("xn--" + Punycode) any
 * label containing a non-ASCII code point, rejoin (empty labels, including a
 * trailing one, pass through as in the TS split('.')). Returns a malloc'd
 * NUL-terminated string (the caller frees) or NULL on allocation failure.
 * Empty input -> "". */
char *punycode_encode(const char *domain) {
    if (!*domain) return strdup("");
    PcStr out = {0};
    char *lower = strdup(domain);
    if (!lower) return NULL;
    for (char *p = lower; *p; p++)
        if (*p >= 'A' && *p <= 'Z') *p += 'a' - 'A';

    bool ok = true, first = true;
    const char *p = lower;
    for (;;) {
        const char *dot = strchr(p, '.');
        size_t ll = dot ? (size_t)(dot - p) : strlen(p);
        if (ok && !first) {
            char d = '.';
            ok = pc_str_push(&out, &d, 1);
        }
        char *lab = strndup(p, ll); /* bound the label: helpers read to NUL */
        if (!lab) { ok = false; break; }
        if (pc_has_non_ascii(lab)) {
            PcStr enc = {0};
            ok = ok && pc_str_push(&out, PC_ACE_PREFIX, 4) &&
                 pc_encode_label(lab, &enc) && pc_str_push(&out, enc.p, enc.len);
            free(enc.p);
        } else {
            ok = ok && pc_str_push(&out, lab, ll);
        }
        free(lab);
        if (!dot || !ok) break;
        first = false;
        p = dot + 1;
    }
    free(lower);
    if (!ok) { free(out.p); return NULL; }
    if (!pc_str_push(&out, "", 1)) { free(out.p); return NULL; } /* NUL */
    return out.p;
}

/* IDNA toUnicode: split on '.', decode any "xn--" label (case-insensitive,
 * prefix detected on the lowercased label), reject the whole domain when one
 * is invalid. Returns a malloc'd NUL-terminated string or NULL (malformed /
 * allocation). Empty input -> "". */
char *punycode_decode(const char *domain) {
    if (!*domain) return strdup("");
    PcStr out = {0};
    bool ok = true, first = true;
    const char *p = domain;
    for (;;) {
        const char *dot = strchr(p, '.');
        size_t ll = dot ? (size_t)(dot - p) : strlen(p);
        if (ok && !first) {
            char d = '.';
            ok = pc_str_push(&out, &d, 1);
        }
        char low[64]; /* lowercased prefix of the label, for the ACE check */
        size_t ml = ll < 63 ? ll : 63;
        for (size_t i = 0; i < ml; i++) {
            char c = p[i];
            low[i] = (c >= 'A' && c <= 'Z') ? (char)(c + 'a' - 'A') : c;
        }
        if (ll > 4 && strncmp(low, PC_ACE_PREFIX, 4) == 0) {
            char *lab = strndup(p + 4, ll - 4);
            PcStr dec = {0};
            ok = ok && lab && pc_decode_label(lab, &dec) &&
                 pc_str_push(&out, dec.p, dec.len);
            free(lab);
            free(dec.p);
        } else {
            ok = ok && pc_str_push(&out, p, ll);
        }
        if (!dot || !ok) break;
        first = false;
        p = dot + 1;
    }
    if (!ok) { free(out.p); return NULL; }
    if (!pc_str_push(&out, "", 1)) { free(out.p); return NULL; } /* NUL */
    return out.p;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →