Skip to content

IPv4 ↔ IPv6 Converter — Zig source

Convert between IPv4 and IPv6 addresses both ways. Parse and validate addresses, expand and compress IPv6 to its canonical RFC 5952 form, map an IPv4 into IPv4-mapped and IPv4-compatible IPv6 (or any custom /96 prefix), and extract an embedded IPv4 back out.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! ip-converter — IPv4 ↔ IPv6 conversion (polyglot showcase: Zig).
//!
//! Language: Zig 0.13 (standard library only)
//! Source:   CosmoDev polyglot showcase port of the ip-converter tool,
//!           ported from src/lib/ip-converter.ts (the canonical TypeScript
//!           implementation).
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Pure, deterministic IPv4/IPv6 address conversion logic. Every parse
//! function returns null (or error.Invalid for the string renderers) on
//! invalid input rather than panicking, so the UI can show a graceful error.
//! IPv6 text follows RFC 5952: lowercase hex, no leading zeros, the single
//! longest run of zero groups collapsed to "::", and a dotted-decimal tail
//! only for IPv4-mapped ("::ffff:") addresses.
//!
//! Zig's types stay precise the way the Rust port's do: octets are u8, IPv6
//! 16-bit groups are u16. Parse functions return optionals; renderers take a
//! caller-supplied allocator and return an error union (error.Invalid for
//! unparseable input, error.OutOfMemory for allocation failure). The TS
//! regex guards are hand-rolled as length + charset checks, mirroring the
//! Rust port.

const std = @import("std");

/// Embedding family for placing an IPv4 quad inside an IPv6 address.
pub const EmbedMode = enum {
    /// `::ffff:a.b.c.d` — the modern, non-deprecated IPv4-mapped form (default).
    mapped,
    /// `::a.b.c.d` — the deprecated IPv4-compatible form.
    compatible,
};

/// Options for embedding an IPv4 octet quad into an IPv6 address.
pub const Ipv4ToIpv6Options = struct {
    /// Embedding family. Ignored when `prefix` is non-null.
    mode: EmbedMode = .mapped,
    /// Optional custom high-96-bit prefix (a valid IPv6 string; its first 6
    /// groups are used and its low 32 bits are overwritten by the IPv4). e.g.
    /// `"64:ff9b::"` yields a NAT64-style `64:ff9b::a.b.c.d`. Overrides `mode`.
    prefix: ?[]const u8 = null,
};

/// Renderer failures: unparseable input, or an allocation error.
pub const Error = error{ Invalid, OutOfMemory };

/// The character set Python's str.strip() removes (all ASCII whitespace).
const whitespace = " \t\n\r\x0b\x0c";

/// A valid IPv6 group token: 1-4 hex digits, no sign, no underscores.
/// (Equivalent to the TS `/^[0-9a-fA-F]{1,4}$/` regex.)
fn isHexGroup(tok: []const u8) bool {
    if (tok.len < 1 or tok.len > 4) return false;
    for (tok) |c| {
        const ok = (c >= '0' and c <= '9') or (c >= 'a' and c <= 'f') or (c >= 'A' and c <= 'F');
        if (!ok) return false;
    }
    return true;
}

/// A valid IPv4 octet token: 1-3 decimal digits. Range is enforced separately.
/// (Equivalent to the TS `/^\d{1,3}$/` regex.)
fn isDec3(tok: []const u8) bool {
    if (tok.len < 1 or tok.len > 3) return false;
    for (tok) |c| {
        if (c < '0' or c > '9') return false;
    }
    return true;
}

/// Count non-overlapping occurrences of "::" — used to enforce the
/// at-most-one-compression rule.
fn countDoubleColon(haystack: []const u8) usize {
    var n: usize = 0;
    var i: usize = 0;
    while (std.mem.indexOfPos(u8, haystack, i, "::")) |pos| {
        n += 1;
        i = pos + 2;
    }
    return n;
}

/// Parse a dotted-decimal IPv4 string into four octets, validating each is
/// 0-255. Returns null for anything that is not exactly four numeric octets
/// in range.
pub fn parseIpv4(s: []const u8) ?[4]u8 {
    const input = std.mem.trim(u8, s, whitespace);
    var it = std.mem.splitScalar(u8, input, '.');
    var octets: [4]u8 = undefined;
    var n: usize = 0;
    while (it.next()) |tok| {
        if (n >= 4) return null; // more than four tokens is invalid
        if (!isDec3(tok)) return null;
        // isDec3 guarantees digits only, so the base-10 parse cannot fail.
        const v = std.fmt.parseInt(u32, tok, 10) catch return null;
        if (v > 255) return null;
        octets[n] = @intCast(v);
        n += 1;
    }
    return if (n == 4) octets else null;
}

/// Parse an IPv6 string (with "::" compression, hex groups, and an optional
/// dotted-decimal IPv4 tail for mapped/compatible forms) into eight 16-bit
/// groups. Returns null on any malformed input — never panics.
pub fn parseIpv6(s: []const u8) ?[8]u16 {
    const input = std.mem.trim(u8, s, whitespace);
    if (input.len == 0) return null;
    // At most one "::" run is legal; reject ambiguous double-compression.
    if (countDoubleColon(input) > 1) return null;

    // Branch on the position of "::" (if any). The two arms mirror each
    // other: split into tokens, validate each, allow a dotted-quad only in
    // the final slot, then assemble exactly eight groups.
    if (std.mem.indexOf(u8, input, "::")) |dc| {
        const before = input[0..dc];
        const after = input[dc + 2 ..];

        var head: [8]u16 = undefined;
        var hn: usize = 0;
        if (before.len > 0) {
            var it = std.mem.splitScalar(u8, before, ':');
            while (it.next()) |tok| {
                // A 9th head group would push the total to >= 8 groups, which
                // the total check below rejects anyway.
                if (hn >= head.len) return null;
                if (!isHexGroup(tok)) return null;
                head[hn] = std.fmt.parseInt(u16, tok, 16) catch return null;
                hn += 1;
            }
        }

        // Collect tail tokens first so the final one (which may be a dotted
        // quad) is knowable, then parse. Any input with 10+ tail tokens can
        // never total fewer than 8 groups.
        var toks: [9][]const u8 = undefined;
        var ntok: usize = 0;
        if (after.len > 0) {
            var it = std.mem.splitScalar(u8, after, ':');
            while (it.next()) |tok| {
                if (ntok >= toks.len) return null;
                toks[ntok] = tok;
                ntok += 1;
            }
        }

        var tail: [8]u16 = undefined;
        var tn: usize = 0;
        for (toks[0..ntok], 0..) |tok, i| {
            // A dotted-quad IPv4 tail is permitted only in the final slot,
            // where it contributes two groups (high octet pair, low octet pair).
            const is_last = i == ntok - 1;
            if (is_last and std.mem.indexOfScalar(u8, tok, '.') != null) {
                const oct = parseIpv4(tok) orelse return null;
                if (tn + 2 > tail.len) return null;
                tail[tn] = (@as(u16, oct[0]) << 8) | oct[1];
                tail[tn + 1] = (@as(u16, oct[2]) << 8) | oct[3];
                tn += 2;
            } else {
                if (!isHexGroup(tok)) return null;
                if (tn + 1 > tail.len) return null;
                tail[tn] = std.fmt.parseInt(u16, tok, 16) catch return null;
                tn += 1;
            }
        }

        const total = hn + tn;
        if (total >= 8) return null; // "::" must elide at least one group.
        var groups = [_]u16{0} ** 8;
        for (head[0..hn], 0..) |v, k| groups[k] = v;
        // The middle [hn .. 8 - tn] stays zero — that is the elided run "::"
        // stands in for.
        for (tail[0..tn], 0..) |v, k| groups[8 - tn + k] = v;
        return groups;
    }

    // No compression: split on ':' and parse, allowing a dotted-quad only in
    // the last slot. The result must be exactly eight groups.
    var toks: [9][]const u8 = undefined;
    var ntok: usize = 0;
    {
        var it = std.mem.splitScalar(u8, input, ':');
        while (it.next()) |tok| {
            if (ntok >= toks.len) return null;
            toks[ntok] = tok;
            ntok += 1;
        }
    }
    var groups = [_]u16{0} ** 8;
    var n: usize = 0;
    for (toks[0..ntok], 0..) |tok, i| {
        const is_last = i == ntok - 1;
        if (is_last and std.mem.indexOfScalar(u8, tok, '.') != null) {
            const oct = parseIpv4(tok) orelse return null;
            if (n + 2 > 8) return null;
            groups[n] = (@as(u16, oct[0]) << 8) | oct[1];
            groups[n + 1] = (@as(u16, oct[2]) << 8) | oct[3];
            n += 2;
        } else {
            if (!isHexGroup(tok)) return null;
            if (n + 1 > 8) return null;
            groups[n] = std.fmt.parseInt(u16, tok, 16) catch return null;
            n += 1;
        }
    }
    return if (n == 8) groups else null;
}

/// True when the eight groups form an IPv4-mapped ("::ffff:") address.
fn isMapped(g: [8]u16) bool {
    return g[0] == 0 and g[1] == 0 and g[2] == 0 and g[3] == 0 and g[4] == 0 and g[5] == 0xffff;
}

/// True when the eight groups form an IPv4-compatible ("::") address.
fn isCompatible(g: [8]u16) bool {
    return g[0] == 0 and g[1] == 0 and g[2] == 0 and g[3] == 0 and g[4] == 0 and g[5] == 0;
}

/// Collapse the longest run (length >= 2) of zero groups into "::" (first run
/// wins on ties) and strip leading zeros — RFC 5952 canonical text for
/// pure-hex IPv6. Works over any group slice (8 for a whole address, 6 for
/// the high part of an embedded-IPv4 render). Does not emit dotted-decimal;
/// call `renderCanonical` for that.
fn compressGroups(writer: anytype, groups: []const u16) !void {
    var best_start: usize = 0;
    var best_len: usize = 0;
    var cur_start: usize = 0;
    var cur_len: usize = 0;
    // Track the longest run of consecutive zero groups. best_start records
    // the first run of the longest length (strict > keeps earliest).
    for (groups, 0..) |v, i| {
        if (v == 0) {
            if (cur_len == 0) cur_start = i;
            cur_len += 1;
            if (cur_len > best_len) {
                best_len = cur_len;
                best_start = cur_start;
            }
        } else {
            cur_len = 0;
        }
    }

    if (best_len < 2) {
        for (groups, 0..) |v, i| {
            if (i > 0) try writer.writeByte(':');
            // "{x}" renders lowercase hex with no leading zeros.
            try writer.print("{x}", .{v});
        }
        return;
    }
    for (groups[0..best_start], 0..) |v, i| {
        if (i > 0) try writer.writeByte(':');
        try writer.print("{x}", .{v});
    }
    try writer.writeAll("::");
    for (groups[best_start + best_len ..], 0..) |v, i| {
        if (i > 0) try writer.writeByte(':');
        try writer.print("{x}", .{v});
    }
}

/// Render four octets as `a.b.c.d`. (The octet type is u8, which already
/// guarantees range, but the signature keeps the symmetry with the other
/// ports.)
pub fn ipv4ToString(allocator: std.mem.Allocator, octets: [4]u8) Error![]u8 {
    return std.fmt.allocPrint(allocator, "{d}.{d}.{d}.{d}", .{
        octets[0], octets[1], octets[2], octets[3],
    });
}

/// Render a compressed high part followed by a dotted-decimal IPv4 tail. When
/// the high part already ends in "::" (its zero run reaches the boundary) the
/// IPv4 attaches directly; otherwise a single ":" separates them — so
/// "::ffff:" → "::ffff:a.b.c.d" and "::" → "::a.b.c.d".
fn renderWithEmbeddedTail(
    out: *std.ArrayList(u8),
    high: []const u16,
    octets: [4]u8,
) std.mem.Allocator.Error![]u8 {
    try compressGroups(out.writer(), high);
    // Decide the separator from what has been written so far, then append.
    const sep: []const u8 = if (std.mem.endsWith(u8, out.items, "::")) "" else ":";
    try out.writer().print("{s}{d}.{d}.{d}.{d}", .{
        sep, octets[0], octets[1], octets[2], octets[3],
    });
    return out.items;
}

/// Canonical RFC 5952 text for eight groups: a dotted-decimal tail for
/// IPv4-mapped ("::ffff:") addresses, otherwise pure compressed hex. The
/// deprecated IPv4-compatible range ("::/96") is NOT rendered dotted here —
/// that would mis-render the unspecified ("::") and loopback ("::1")
/// addresses as "::0.0.0.0" / "::0.0.0.1". Compatible extraction is still
/// available via `ipv6ToIpv4`; on-demand compatible generation via
/// `ipv4ToIpv6` is untouched.
fn renderCanonical(out: *std.ArrayList(u8), groups: [8]u16) std.mem.Allocator.Error![]u8 {
    if (isMapped(groups)) {
        const octets = [4]u8{
            @intCast(groups[6] >> 8),
            @intCast(groups[6] & 0xff),
            @intCast(groups[7] >> 8),
            @intCast(groups[7] & 0xff),
        };
        return renderWithEmbeddedTail(out, groups[0..6], octets);
    }
    try compressGroups(out.writer(), groups[0..]);
    return out.items;
}

/// Render eight groups as canonical compressed IPv6.
pub fn ipv6ToString(allocator: std.mem.Allocator, groups: [8]u16) Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    return renderCanonical(&out, groups);
}

/// Expand an IPv6 string to its full eight-group, four-hex-digit form;
/// error.Invalid if invalid.
pub fn expandIpv6(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    const g = parseIpv6(s) orelse return error.Invalid;
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    const w = out.writer();
    for (g, 0..) |v, i| {
        if (i > 0) try w.writeByte(':');
        // "{x:0>4}" left-pads each group to a fixed 4-digit width: 0000..ffff.
        try w.print("{x:0>4}", .{v});
    }
    return out.toOwnedSlice();
}

/// Compress an IPv6 string to its RFC 5952 canonical form; error.Invalid if
/// invalid.
pub fn compressIpv6(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    const g = parseIpv6(s) orelse return error.Invalid;
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    return renderCanonical(&out, g);
}

/// Embed an IPv4 octet quad into an IPv6 address. By default produces the
/// IPv4-mapped form "::ffff:a.b.c.d"; `.compatible` yields "::a.b.c.d"; a set
/// `opts.prefix` overrides both and places the IPv4 after any custom /96
/// prefix (e.g. "64:ff9b::a.b.c.d"). Returns error.Invalid for an invalid
/// prefix.
pub fn ipv4ToIpv6(
    allocator: std.mem.Allocator,
    octets: [4]u8,
    opts: Ipv4ToIpv6Options,
) Error![]u8 {
    var out = std.ArrayList(u8).init(allocator);
    errdefer out.deinit();
    if (opts.prefix) |prefix| {
        const p = parseIpv6(prefix) orelse return error.Invalid;
        return renderWithEmbeddedTail(&out, p[0..6], octets);
    }
    const high: [6]u16 = switch (opts.mode) {
        .compatible => .{ 0, 0, 0, 0, 0, 0 },
        .mapped => .{ 0, 0, 0, 0, 0, 0xffff },
    };
    return renderWithEmbeddedTail(&out, high[0..], octets);
}

/// Extract the embedded IPv4 from an IPv4-mapped ("::ffff:a.b.c.d") or
/// IPv4-compatible ("::a.b.c.d") address, returning dotted-decimal or
/// error.Invalid when the address carries no embedded IPv4 (or is
/// unparseable).
pub fn ipv6ToIpv4(allocator: std.mem.Allocator, s: []const u8) Error![]u8 {
    const g = parseIpv6(s) orelse return error.Invalid;
    if (!(isMapped(g) or isCompatible(g))) return error.Invalid;
    const octets = [4]u8{
        @intCast(g[6] >> 8),
        @intCast(g[6] & 0xff),
        @intCast(g[7] >> 8),
        @intCast(g[7] & 0xff),
    };
    return ipv4ToString(allocator, octets);
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →