Skip to content

Punycode Converter — Zig source

Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//
// Language:   Zig (0.13+, standard library only)
// Source:     CosmoDev polyglot showcase port of the Punycode tool, ported from
//             src/lib/punycode.ts (the canonical TypeScript implementation) and
//             cli/punycode/punycode.go (the live Go CLI twin).
// License:    display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; encode never fails (only allocation can), decode
//     rejects malformed input with null.
//   - Functionally equivalent to the TS/Go reference: same inputs -> same
//     outputs.
//   - Self-contained: stdlib only. Strings are UTF-8 slices; std.unicode
//     iterators/encoders map the code points, so astral characters (emoji,
//     CJK extensions) are single elements, matching Go runes and the TS
//     code-point iteration. i64 carries the RFC 3492 arithmetic with the
//     2^53-1 (Number.MAX_SAFE_INTEGER) overflow guard; code points beyond
//     U+10FFFF reject the label. Lowercasing is ASCII-only (std.ascii),
//     which covers IDNA host-label syntax.

const std = @import("std");

const base: i64 = 36;
const tmin: i64 = 1;
const tmax: i64 = 26;
const skew: i64 = 38;
const damp: i64 = 700;
const initial_bias: i64 = 72;
const initial_n: i64 = 128;
const ace_prefix = "xn--";
const max_int: i64 = 9007199254740991; // 2^53-1 overflow guard

/// Bias adaptation (RFC 3492 section 6.1).
fn adapt(delta: i64, numpoints: i64, firsttime: bool) i64 {
    var d: i64 = if (firsttime) @divTrunc(delta, damp) else @divTrunc(delta, 2);
    d += @divTrunc(d, numpoints);
    var k: i64 = 0;
    while (d > @divTrunc((base - tmin) * tmax, 2)) {
        d = @divTrunc(d, base - tmin);
        k += base;
    }
    return k + @divTrunc((base - tmin + 1) * d, d + skew);
}

/// Map a digit value (0-35) to its base-36 character (lowercase).
fn digitToChar(d: i64) u8 {
    return if (d < 26) @intCast(97 + d) else @intCast(48 + (d - 26));
}

/// Map a byte to its digit value (0-35), case-insensitive, or -1 when invalid.
fn charToDigit(c: u8) i64 {
    if (c >= 'a' and c <= 'z') return c - 'a';
    if (c >= 'A' and c <= 'Z') return c - 'A';
    if (c >= '0' and c <= '9') return c - '0' + 26;
    return -1;
}

/// True if the UTF-8 slice contains any non-ASCII code point (>= 0x80).
fn hasNonAscii(s: []const u8) bool {
    for (s) |c| {
        if (c >= 128) return true;
    }
    return false;
}

/// Append code point `cp` as UTF-8 to `out`.
fn appendCodePoint(out: *std.ArrayList(u8), cp: i64) !void {
    var buf: [4]u8 = undefined;
    const n = std.unicode.utf8Encode(@intCast(cp), &buf) catch return error.InvalidCodePoint;
    try out.appendSlice(buf[0..n]);
}

/// Punycode-encode a single label (RFC 3492), no ACE prefix. Basic code
/// points are emitted first, then a '-' delimiter (only if there was at least
/// one), then the generalized-base-36 deltas. Caller owns the result.
pub fn encodeLabel(a: std.mem.Allocator, input: []const u8) ![]u8 {
    var cps = std.ArrayList(i64).init(a);
    defer cps.deinit();
    var it = (try std.unicode.Utf8View.init(input)).iterator();
    while (it.nextCodepoint()) |cp| try cps.append(cp); // astral chars are one element

    const length: i64 = @intCast(cps.items.len);
    var out = std.ArrayList(u8).init(a);
    errdefer out.deinit();

    var b: i64 = 0;
    for (cps.items) |cp| {
        if (cp < 128) {
            try out.append(@intCast(cp));
            b += 1;
        }
    }
    if (b > 0) try out.append('-');

    var n = initial_n;
    var delta: i64 = 0;
    var bias = initial_bias;
    var h = b;
    while (h < length) {
        var m: i64 = std.math.maxInt(i64); // smallest code point in the input that is >= n
        for (cps.items) |cp| {
            if (cp >= n and cp < m) m = cp;
        }
        delta += (m - n) * (h + 1);
        n = m;
        for (cps.items) |cp| {
            if (cp < n) {
                delta += 1;
            } else if (cp == n) {
                var q = delta;
                var k = base;
                while (true) {
                    const t = @max(tmin, @min(tmax, k - bias));
                    if (q < t) break;
                    try out.append(digitToChar(t + @rem(q - t, base - t)));
                    q = @divTrunc(q - t, base - t);
                    k += base;
                }
                try out.append(digitToChar(q));
                bias = adapt(delta, h + 1, h == b);
                delta = 0;
                h += 1;
            }
        }
        delta += 1;
        n += 1;
    }

    return out.toOwnedSlice();
}

/// Punycode-decode a single label (RFC 3492). Returns null when the input is
/// malformed (invalid digit, truncated generalized number, non-ASCII in the
/// basic portion, code point beyond U+10FFFF, or overflow). Caller owns the
/// result.
pub fn decodeLabel(a: std.mem.Allocator, input: []const u8) !?[]u8 {
    const last_dash = std.mem.lastIndexOfScalar(u8, input, '-');
    var out = std.ArrayList(i64).init(a); // decoded code points
    defer out.deinit();

    if (last_dash) |dash| {
        for (input[0..dash]) |c| {
            if (c >= 128) return null; // basic portion must be ASCII
            try out.append(c);
        }
    }
    const ext = if (last_dash) |dash| input[dash + 1 ..] else input;

    var n = initial_n;
    var i: i64 = 0;
    var bias = initial_bias;
    var pos: usize = 0;
    while (pos < ext.len) {
        const oldi = i;
        var w: i64 = 1;
        var k = base;
        while (true) {
            if (pos >= ext.len) return null; // truncated generalized number
            const digit = charToDigit(ext[pos]);
            if (digit < 0) return null; // invalid digit
            pos += 1;
            if (digit >= @divTrunc(max_int, w)) return null; // overflow guard
            i += digit * w;
            const t = @max(tmin, @min(tmax, k - bias));
            if (digit < t) break;
            w *= base - t;
            k += base;
        }
        bias = adapt(i - oldi, @as(i64, @intCast(out.items.len)) + 1, oldi == 0);
        const out_len = @as(i64, @intCast(out.items.len)) + 1;
        n += @divTrunc(i, out_len);
        i = @rem(i, out_len);
        if (n < 0 or n > 0x10FFFF) return null;
        try out.insert(@intCast(i), n);
        i += 1;
    }

    var result = std.ArrayList(u8).init(a);
    errdefer result.deinit();
    for (out.items) |cp| try appendCodePoint(&result, cp);
    return try result.toOwnedSlice();
}

/// Split on '.', keeping empty labels (including a trailing one), like the TS
/// String.split('.').
fn splitLabels(a: std.mem.Allocator, s: []const u8) ![][]const u8 {
    var labels = std.ArrayList([]const u8).init(a);
    errdefer labels.deinit();
    var start: usize = 0;
    for (s, 0..) |c, idx| {
        if (c == '.') {
            try labels.append(s[start..idx]);
            start = idx + 1;
        }
    }
    try labels.append(s[start..]);
    return labels.toOwnedSlice();
}

/// IDNA toASCII: lowercase the domain (ASCII), ACE-encode ("xn--" +
/// Punycode) any label containing a non-ASCII code point, leave ASCII-only
/// labels untouched. Empty input returns an owned empty string. Caller owns
/// the result.
pub fn encode(a: std.mem.Allocator, domain: []const u8) ![]u8 {
    if (domain.len == 0) return try a.dupe(u8, "");

    var lower = try a.dupe(u8, domain);
    defer a.free(lower);
    for (lower) |*c| c.* = std.ascii.toLower(c.*);

    const labels = try splitLabels(a, lower);
    defer a.free(labels);

    var out = std.ArrayList(u8).init(a);
    errdefer out.deinit();
    for (labels, 0..) |label, idx| {
        if (idx > 0) try out.append('.');
        if (hasNonAscii(label)) {
            try out.appendSlice(ace_prefix);
            const enc = try encodeLabel(a, label);
            defer a.free(enc);
            try out.appendSlice(enc);
        } else {
            try out.appendSlice(label);
        }
    }
    return out.toOwnedSlice();
}

/// IDNA toUnicode: decode any "xn--" label (case-insensitive, prefix detected
/// on the lowercased label), leave every other label untouched. Returns null
/// when any "xn--" label is invalid - the whole domain is rejected, matching
/// IDNA semantics. Empty input returns an owned empty string. Caller owns the
/// result.
pub fn decode(a: std.mem.Allocator, domain: []const u8) !?[]u8 {
    if (domain.len == 0) return try a.dupe(u8, "");

    const labels = try splitLabels(a, domain); // slices into `domain` - not individually owned
    defer a.free(labels);

    var out = std.ArrayList(u8).init(a);
    errdefer out.deinit();
    for (labels, 0..) |label, idx| {
        if (idx > 0) try out.append('.');
        var low_buf: [4]u8 = undefined;
        const low = blk: { // lowercase just the 4-byte prefix for the ACE check
            const m = @min(label.len, ace_prefix.len);
            for (label[0..m], 0..) |c, j| low_buf[j] = std.ascii.toLower(c);
            break :blk low_buf[0..m];
        };
        if (label.len > ace_prefix.len and std.mem.eql(u8, low, ace_prefix)) {
            const dec = try decodeLabel(a, label[ace_prefix.len..]);
            if (dec == null) return null; // reject the whole domain on any invalid label
            defer a.free(dec.?);
            try out.appendSlice(dec.?);
        } else {
            try out.appendSlice(label);
        }
    }
    return try out.toOwnedSlice();
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →