Skip to content

Base64 Encode / Decode — Zig source

Encode text to Base64 or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! base64 — UTF-8 safe Base64 encode/decode.
//!
//! Language: Zig (0.13, standard library only — std.base64 can do the
//!           packing, but the codec is hand-rolled like rust.rs so the
//!           strict validation contract stays visible in one place)
//! Source:   CosmoDev polyglot showcase port of the `base64` tool, ported
//!           from src/lib/base64.ts (the canonical TypeScript implementation);
//!           algorithm and structure mirror src/tool-sources/base64/rust.rs.
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Slices are byte sequences, so a `[]const u8` holding UTF-8 text is already
//! the byte sequence Base64 operates on — there is no separate "encode to
//! UTF-8" step, unlike the TypeScript port's TextEncoder. On decode the
//! output is validated with std.unicode.utf8ValidateSlice before being
//! returned — the TextDecoder step.
//!
//! Test: zig test zig.zig

const std = @import("std");

/// Standard Base64 alphabet (RFC 4648). The index of each byte is its 6-bit
/// value — the same alphabet btoa emits in the browser.
const alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";

/// Sentinel for "this byte is not part of the Base64 alphabet".
const invalid: i8 = -1;
/// Sentinel for "this byte is the '=' padding character".
const padding: i8 = -2;

pub const DecodeError = error{
    /// A byte outside the alphabet (and outside padding) appeared in the input.
    IllegalCharacter,
    /// After whitespace stripping, the length is not a multiple of 4.
    BadLength,
    /// The decoded bytes are not valid UTF-8.
    InvalidUtf8,
    /// Allocation failure.
    OutOfMemory,
};

/// 256-entry lookup table mapping an ASCII byte to its 6-bit value or one of
/// the sentinels. Built per decode call — 256 writes is cheaper in Zig than
/// the machinery to cache it.
fn decodeTable() [256]i8 {
    var table = [_]i8{invalid} ** 256;
    for (alphabet, 0..) |byte, value| table[byte] = @intCast(value);
    table['='] = padding;
    return table;
}

/// Encode UTF-8 text into standard, padded Base64.
///
/// Walks the bytes in 3-byte groups, emitting four 6-bit indices per group.
/// A trailing partial group (1 or 2 bytes) is padded with '=' so the output
/// length is always a multiple of 4 — the same shape as btoa in the browser.
/// The caller owns the returned slice.
pub fn b64Encode(allocator: std.mem.Allocator, input: []const u8) error{OutOfMemory}![]u8 {
    const out = try allocator.alloc(u8, (input.len + 2) / 3 * 4);
    var o: usize = 0;
    var i: usize = 0;

    // Complete 3-byte chunks -> four Base64 characters.
    while (i + 3 <= input.len) : (i += 3) {
        const triple = (@as(usize, input[i]) << 16) |
            (@as(usize, input[i + 1]) << 8) | @as(usize, input[i + 2]);
        out[o] = alphabet[(triple >> 18) & 0x3F];
        out[o + 1] = alphabet[(triple >> 12) & 0x3F];
        out[o + 2] = alphabet[(triple >> 6) & 0x3F];
        out[o + 3] = alphabet[triple & 0x3F];
        o += 4;
    }

    // Trailing 1 or 2 bytes, padded so the output stays a multiple of 4.
    const remainder = input.len - i;
    if (remainder == 1) {
        const triple = @as(usize, input[i]) << 16;
        out[o] = alphabet[(triple >> 18) & 0x3F];
        out[o + 1] = alphabet[(triple >> 12) & 0x3F];
        out[o + 2] = '=';
        out[o + 3] = '=';
    } else if (remainder == 2) {
        const triple = (@as(usize, input[i]) << 16) |
            (@as(usize, input[i + 1]) << 8);
        out[o] = alphabet[(triple >> 18) & 0x3F];
        out[o + 1] = alphabet[(triple >> 12) & 0x3F];
        out[o + 2] = alphabet[(triple >> 6) & 0x3F];
        out[o + 3] = '=';
    }

    return out;
}

/// Decode standard Base64 back into the original UTF-8 text.
///
/// Whitespace inside the input is stripped first (std.ascii.isWhitespace
/// matches the six characters JavaScript's \s recognizes), so line-wrapped
/// Base64 decodes cleanly. Any malformed input — an illegal character, a
/// length that is not a multiple of 4, or decoded bytes that are not valid
/// UTF-8 — returns an error, matching the TS port's "throw on invalid input"
/// contract. The caller owns the returned slice.
pub fn b64Decode(allocator: std.mem.Allocator, input: []const u8) DecodeError![]u8 {
    // Drop every whitespace byte, keeping the surviving ASCII bytes.
    const cleaned = try allocator.alloc(u8, input.len);
    defer allocator.free(cleaned);
    var n: usize = 0;
    for (input) |byte| {
        if (!std.ascii.isWhitespace(byte)) {
            cleaned[n] = byte;
            n += 1;
        }
    }

    // Standard Base64 with padding is always a multiple of 4 characters.
    if (n % 4 != 0) return error.BadLength;

    // Count trailing '=' padding (0, 1, or 2 in well-formed input).
    var pad: usize = 0;
    while (pad < 2 and n > pad and cleaned[n - 1 - pad] == '=') pad += 1;

    // Every group yields 3 bytes, minus what the final group's padding holds back.
    const bytes = try allocator.alloc(u8, n / 4 * 3 - pad);
    var o: usize = 0;
    var i: usize = 0;
    const table = decodeTable();

    while (i < n) : (i += 4) {
        // Read four sextets, validating each against the lookup table.
        var sextets = [4]u8{ 0, 0, 0, 0 };
        for (0..4) |j| {
            const entry = table[cleaned[i + j]];
            if (entry == padding) {
                sextets[j] = 0; // padding contributes zero bits
            } else if (entry == invalid) {
                allocator.free(bytes);
                return error.IllegalCharacter;
            } else {
                sextets[j] = @intCast(entry);
            }
        }

        const triple = (@as(usize, sextets[0]) << 18) |
            (@as(usize, sextets[1]) << 12) |
            (@as(usize, sextets[2]) << 6) | @as(usize, sextets[3]);
        const is_last_group = i + 4 == n;

        bytes[o] = @truncate(triple >> 16); // byte 0: always present
        o += 1;
        if (!(is_last_group and pad == 2)) { // byte 1: absent only in a 2-pad final group
            bytes[o] = @truncate(triple >> 8);
            o += 1;
        }
        if (!(is_last_group and pad >= 1)) { // byte 2: absent whenever there is any padding
            bytes[o] = @truncate(triple);
            o += 1;
        }
    }

    // Re-interpret the decoded bytes as UTF-8 (the TextDecoder step).
    if (!std.unicode.utf8ValidateSlice(bytes)) {
        allocator.free(bytes);
        return error.InvalidUtf8;
    }
    return bytes;
}

test "encode matches the known vectors" {
    const t = std.testing;
    const a = t.allocator;

    const hello = try b64Encode(a, "Hello, world!");
    defer a.free(hello);
    try t.expectEqualStrings("SGVsbG8sIHdvcmxkIQ==", hello);

    const one = try b64Encode(a, "a");
    defer a.free(one);
    try t.expectEqualStrings("YQ==", one);

    const three = try b64Encode(a, "abc");
    defer a.free(three);
    try t.expectEqualStrings("YWJj", three);
}

test "decode strips whitespace and round-trips UTF-8" {
    const t = std.testing;
    const a = t.allocator;

    const text = try b64Decode(a, "aGVs\nbG8g d29ybGQ=");
    defer a.free(text);
    try t.expectEqualStrings("hello world", text);

    const unicode = "héllo 🌍";
    const encoded = try b64Encode(a, unicode);
    defer a.free(encoded);
    const decoded = try b64Decode(a, encoded);
    defer a.free(decoded);
    try t.expectEqualStrings(unicode, decoded);
}

test "decode rejects malformed input" {
    const t = std.testing;
    const a = t.allocator;

    try t.expectError(error.BadLength, b64Decode(a, "SGVsbG8"));
    try t.expectError(error.IllegalCharacter, b64Decode(a, "SGVsbG8*"));
    // "/w==" decodes to the single byte 0xFF, which is not valid UTF-8.
    try t.expectError(error.InvalidUtf8, b64Decode(a, "/w=="));
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →