Punycode Converter — Zig source
Convert internationalized domain names (IDN) between Unicode and Punycode (xn--) ACE form. RFC 3492 compliant, runs entirely in your browser, with a shareable link to your exact input.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// punycode - RFC 3492 Punycode encode/decode + IDNA2003 toASCII/toUnicode.
//
// Language: Zig (0.13+, standard library only)
// Source: CosmoDev polyglot showcase port of the Punycode tool, ported from
// src/lib/punycode.ts (the canonical TypeScript implementation) and
// cli/punycode/punycode.go (the live Go CLI twin).
// License: display source - part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; encode never fails (only allocation can), decode
// rejects malformed input with null.
// - Functionally equivalent to the TS/Go reference: same inputs -> same
// outputs.
// - Self-contained: stdlib only. Strings are UTF-8 slices; std.unicode
// iterators/encoders map the code points, so astral characters (emoji,
// CJK extensions) are single elements, matching Go runes and the TS
// code-point iteration. i64 carries the RFC 3492 arithmetic with the
// 2^53-1 (Number.MAX_SAFE_INTEGER) overflow guard; code points beyond
// U+10FFFF reject the label. Lowercasing is ASCII-only (std.ascii),
// which covers IDNA host-label syntax.
const std = @import("std");
const base: i64 = 36;
const tmin: i64 = 1;
const tmax: i64 = 26;
const skew: i64 = 38;
const damp: i64 = 700;
const initial_bias: i64 = 72;
const initial_n: i64 = 128;
const ace_prefix = "xn--";
const max_int: i64 = 9007199254740991; // 2^53-1 overflow guard
/// Bias adaptation (RFC 3492 section 6.1).
fn adapt(delta: i64, numpoints: i64, firsttime: bool) i64 {
var d: i64 = if (firsttime) @divTrunc(delta, damp) else @divTrunc(delta, 2);
d += @divTrunc(d, numpoints);
var k: i64 = 0;
while (d > @divTrunc((base - tmin) * tmax, 2)) {
d = @divTrunc(d, base - tmin);
k += base;
}
return k + @divTrunc((base - tmin + 1) * d, d + skew);
}
/// Map a digit value (0-35) to its base-36 character (lowercase).
fn digitToChar(d: i64) u8 {
return if (d < 26) @intCast(97 + d) else @intCast(48 + (d - 26));
}
/// Map a byte to its digit value (0-35), case-insensitive, or -1 when invalid.
fn charToDigit(c: u8) i64 {
if (c >= 'a' and c <= 'z') return c - 'a';
if (c >= 'A' and c <= 'Z') return c - 'A';
if (c >= '0' and c <= '9') return c - '0' + 26;
return -1;
}
/// True if the UTF-8 slice contains any non-ASCII code point (>= 0x80).
fn hasNonAscii(s: []const u8) bool {
for (s) |c| {
if (c >= 128) return true;
}
return false;
}
/// Append code point `cp` as UTF-8 to `out`.
fn appendCodePoint(out: *std.ArrayList(u8), cp: i64) !void {
var buf: [4]u8 = undefined;
const n = std.unicode.utf8Encode(@intCast(cp), &buf) catch return error.InvalidCodePoint;
try out.appendSlice(buf[0..n]);
}
/// Punycode-encode a single label (RFC 3492), no ACE prefix. Basic code
/// points are emitted first, then a '-' delimiter (only if there was at least
/// one), then the generalized-base-36 deltas. Caller owns the result.
pub fn encodeLabel(a: std.mem.Allocator, input: []const u8) ![]u8 {
var cps = std.ArrayList(i64).init(a);
defer cps.deinit();
var it = (try std.unicode.Utf8View.init(input)).iterator();
while (it.nextCodepoint()) |cp| try cps.append(cp); // astral chars are one element
const length: i64 = @intCast(cps.items.len);
var out = std.ArrayList(u8).init(a);
errdefer out.deinit();
var b: i64 = 0;
for (cps.items) |cp| {
if (cp < 128) {
try out.append(@intCast(cp));
b += 1;
}
}
if (b > 0) try out.append('-');
var n = initial_n;
var delta: i64 = 0;
var bias = initial_bias;
var h = b;
while (h < length) {
var m: i64 = std.math.maxInt(i64); // smallest code point in the input that is >= n
for (cps.items) |cp| {
if (cp >= n and cp < m) m = cp;
}
delta += (m - n) * (h + 1);
n = m;
for (cps.items) |cp| {
if (cp < n) {
delta += 1;
} else if (cp == n) {
var q = delta;
var k = base;
while (true) {
const t = @max(tmin, @min(tmax, k - bias));
if (q < t) break;
try out.append(digitToChar(t + @rem(q - t, base - t)));
q = @divTrunc(q - t, base - t);
k += base;
}
try out.append(digitToChar(q));
bias = adapt(delta, h + 1, h == b);
delta = 0;
h += 1;
}
}
delta += 1;
n += 1;
}
return out.toOwnedSlice();
}
/// Punycode-decode a single label (RFC 3492). Returns null when the input is
/// malformed (invalid digit, truncated generalized number, non-ASCII in the
/// basic portion, code point beyond U+10FFFF, or overflow). Caller owns the
/// result.
pub fn decodeLabel(a: std.mem.Allocator, input: []const u8) !?[]u8 {
const last_dash = std.mem.lastIndexOfScalar(u8, input, '-');
var out = std.ArrayList(i64).init(a); // decoded code points
defer out.deinit();
if (last_dash) |dash| {
for (input[0..dash]) |c| {
if (c >= 128) return null; // basic portion must be ASCII
try out.append(c);
}
}
const ext = if (last_dash) |dash| input[dash + 1 ..] else input;
var n = initial_n;
var i: i64 = 0;
var bias = initial_bias;
var pos: usize = 0;
while (pos < ext.len) {
const oldi = i;
var w: i64 = 1;
var k = base;
while (true) {
if (pos >= ext.len) return null; // truncated generalized number
const digit = charToDigit(ext[pos]);
if (digit < 0) return null; // invalid digit
pos += 1;
if (digit >= @divTrunc(max_int, w)) return null; // overflow guard
i += digit * w;
const t = @max(tmin, @min(tmax, k - bias));
if (digit < t) break;
w *= base - t;
k += base;
}
bias = adapt(i - oldi, @as(i64, @intCast(out.items.len)) + 1, oldi == 0);
const out_len = @as(i64, @intCast(out.items.len)) + 1;
n += @divTrunc(i, out_len);
i = @rem(i, out_len);
if (n < 0 or n > 0x10FFFF) return null;
try out.insert(@intCast(i), n);
i += 1;
}
var result = std.ArrayList(u8).init(a);
errdefer result.deinit();
for (out.items) |cp| try appendCodePoint(&result, cp);
return try result.toOwnedSlice();
}
/// Split on '.', keeping empty labels (including a trailing one), like the TS
/// String.split('.').
fn splitLabels(a: std.mem.Allocator, s: []const u8) ![][]const u8 {
var labels = std.ArrayList([]const u8).init(a);
errdefer labels.deinit();
var start: usize = 0;
for (s, 0..) |c, idx| {
if (c == '.') {
try labels.append(s[start..idx]);
start = idx + 1;
}
}
try labels.append(s[start..]);
return labels.toOwnedSlice();
}
/// IDNA toASCII: lowercase the domain (ASCII), ACE-encode ("xn--" +
/// Punycode) any label containing a non-ASCII code point, leave ASCII-only
/// labels untouched. Empty input returns an owned empty string. Caller owns
/// the result.
pub fn encode(a: std.mem.Allocator, domain: []const u8) ![]u8 {
if (domain.len == 0) return try a.dupe(u8, "");
var lower = try a.dupe(u8, domain);
defer a.free(lower);
for (lower) |*c| c.* = std.ascii.toLower(c.*);
const labels = try splitLabels(a, lower);
defer a.free(labels);
var out = std.ArrayList(u8).init(a);
errdefer out.deinit();
for (labels, 0..) |label, idx| {
if (idx > 0) try out.append('.');
if (hasNonAscii(label)) {
try out.appendSlice(ace_prefix);
const enc = try encodeLabel(a, label);
defer a.free(enc);
try out.appendSlice(enc);
} else {
try out.appendSlice(label);
}
}
return out.toOwnedSlice();
}
/// IDNA toUnicode: decode any "xn--" label (case-insensitive, prefix detected
/// on the lowercased label), leave every other label untouched. Returns null
/// when any "xn--" label is invalid - the whole domain is rejected, matching
/// IDNA semantics. Empty input returns an owned empty string. Caller owns the
/// result.
pub fn decode(a: std.mem.Allocator, domain: []const u8) !?[]u8 {
if (domain.len == 0) return try a.dupe(u8, "");
const labels = try splitLabels(a, domain); // slices into `domain` - not individually owned
defer a.free(labels);
var out = std.ArrayList(u8).init(a);
errdefer out.deinit();
for (labels, 0..) |label, idx| {
if (idx > 0) try out.append('.');
var low_buf: [4]u8 = undefined;
const low = blk: { // lowercase just the 4-byte prefix for the ACE check
const m = @min(label.len, ace_prefix.len);
for (label[0..m], 0..) |c, j| low_buf[j] = std.ascii.toLower(c);
break :blk low_buf[0..m];
};
if (label.len > ace_prefix.len and std.mem.eql(u8, low, ace_prefix)) {
const dec = try decodeLabel(a, label[ace_prefix.len..]);
if (dec == null) return null; // reject the whole domain on any invalid label
defer a.free(dec.?);
try out.appendSlice(dec.?);
} else {
try out.appendSlice(label);
}
}
return try out.toOwnedSlice();
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →