Base64 Encode / Decode — Zig source
Encode text to Base64 or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! base64 — UTF-8 safe Base64 encode/decode.
//!
//! Language: Zig (0.13, standard library only — std.base64 can do the
//! packing, but the codec is hand-rolled like rust.rs so the
//! strict validation contract stays visible in one place)
//! Source: CosmoDev polyglot showcase port of the `base64` tool, ported
//! from src/lib/base64.ts (the canonical TypeScript implementation);
//! algorithm and structure mirror src/tool-sources/base64/rust.rs.
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Slices are byte sequences, so a `[]const u8` holding UTF-8 text is already
//! the byte sequence Base64 operates on — there is no separate "encode to
//! UTF-8" step, unlike the TypeScript port's TextEncoder. On decode the
//! output is validated with std.unicode.utf8ValidateSlice before being
//! returned — the TextDecoder step.
//!
//! Test: zig test zig.zig
const std = @import("std");
/// Standard Base64 alphabet (RFC 4648). The index of each byte is its 6-bit
/// value — the same alphabet btoa emits in the browser.
const alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";
/// Sentinel for "this byte is not part of the Base64 alphabet".
const invalid: i8 = -1;
/// Sentinel for "this byte is the '=' padding character".
const padding: i8 = -2;
pub const DecodeError = error{
/// A byte outside the alphabet (and outside padding) appeared in the input.
IllegalCharacter,
/// After whitespace stripping, the length is not a multiple of 4.
BadLength,
/// The decoded bytes are not valid UTF-8.
InvalidUtf8,
/// Allocation failure.
OutOfMemory,
};
/// 256-entry lookup table mapping an ASCII byte to its 6-bit value or one of
/// the sentinels. Built per decode call — 256 writes is cheaper in Zig than
/// the machinery to cache it.
fn decodeTable() [256]i8 {
var table = [_]i8{invalid} ** 256;
for (alphabet, 0..) |byte, value| table[byte] = @intCast(value);
table['='] = padding;
return table;
}
/// Encode UTF-8 text into standard, padded Base64.
///
/// Walks the bytes in 3-byte groups, emitting four 6-bit indices per group.
/// A trailing partial group (1 or 2 bytes) is padded with '=' so the output
/// length is always a multiple of 4 — the same shape as btoa in the browser.
/// The caller owns the returned slice.
pub fn b64Encode(allocator: std.mem.Allocator, input: []const u8) error{OutOfMemory}![]u8 {
const out = try allocator.alloc(u8, (input.len + 2) / 3 * 4);
var o: usize = 0;
var i: usize = 0;
// Complete 3-byte chunks -> four Base64 characters.
while (i + 3 <= input.len) : (i += 3) {
const triple = (@as(usize, input[i]) << 16) |
(@as(usize, input[i + 1]) << 8) | @as(usize, input[i + 2]);
out[o] = alphabet[(triple >> 18) & 0x3F];
out[o + 1] = alphabet[(triple >> 12) & 0x3F];
out[o + 2] = alphabet[(triple >> 6) & 0x3F];
out[o + 3] = alphabet[triple & 0x3F];
o += 4;
}
// Trailing 1 or 2 bytes, padded so the output stays a multiple of 4.
const remainder = input.len - i;
if (remainder == 1) {
const triple = @as(usize, input[i]) << 16;
out[o] = alphabet[(triple >> 18) & 0x3F];
out[o + 1] = alphabet[(triple >> 12) & 0x3F];
out[o + 2] = '=';
out[o + 3] = '=';
} else if (remainder == 2) {
const triple = (@as(usize, input[i]) << 16) |
(@as(usize, input[i + 1]) << 8);
out[o] = alphabet[(triple >> 18) & 0x3F];
out[o + 1] = alphabet[(triple >> 12) & 0x3F];
out[o + 2] = alphabet[(triple >> 6) & 0x3F];
out[o + 3] = '=';
}
return out;
}
/// Decode standard Base64 back into the original UTF-8 text.
///
/// Whitespace inside the input is stripped first (std.ascii.isWhitespace
/// matches the six characters JavaScript's \s recognizes), so line-wrapped
/// Base64 decodes cleanly. Any malformed input — an illegal character, a
/// length that is not a multiple of 4, or decoded bytes that are not valid
/// UTF-8 — returns an error, matching the TS port's "throw on invalid input"
/// contract. The caller owns the returned slice.
pub fn b64Decode(allocator: std.mem.Allocator, input: []const u8) DecodeError![]u8 {
// Drop every whitespace byte, keeping the surviving ASCII bytes.
const cleaned = try allocator.alloc(u8, input.len);
defer allocator.free(cleaned);
var n: usize = 0;
for (input) |byte| {
if (!std.ascii.isWhitespace(byte)) {
cleaned[n] = byte;
n += 1;
}
}
// Standard Base64 with padding is always a multiple of 4 characters.
if (n % 4 != 0) return error.BadLength;
// Count trailing '=' padding (0, 1, or 2 in well-formed input).
var pad: usize = 0;
while (pad < 2 and n > pad and cleaned[n - 1 - pad] == '=') pad += 1;
// Every group yields 3 bytes, minus what the final group's padding holds back.
const bytes = try allocator.alloc(u8, n / 4 * 3 - pad);
var o: usize = 0;
var i: usize = 0;
const table = decodeTable();
while (i < n) : (i += 4) {
// Read four sextets, validating each against the lookup table.
var sextets = [4]u8{ 0, 0, 0, 0 };
for (0..4) |j| {
const entry = table[cleaned[i + j]];
if (entry == padding) {
sextets[j] = 0; // padding contributes zero bits
} else if (entry == invalid) {
allocator.free(bytes);
return error.IllegalCharacter;
} else {
sextets[j] = @intCast(entry);
}
}
const triple = (@as(usize, sextets[0]) << 18) |
(@as(usize, sextets[1]) << 12) |
(@as(usize, sextets[2]) << 6) | @as(usize, sextets[3]);
const is_last_group = i + 4 == n;
bytes[o] = @truncate(triple >> 16); // byte 0: always present
o += 1;
if (!(is_last_group and pad == 2)) { // byte 1: absent only in a 2-pad final group
bytes[o] = @truncate(triple >> 8);
o += 1;
}
if (!(is_last_group and pad >= 1)) { // byte 2: absent whenever there is any padding
bytes[o] = @truncate(triple);
o += 1;
}
}
// Re-interpret the decoded bytes as UTF-8 (the TextDecoder step).
if (!std.unicode.utf8ValidateSlice(bytes)) {
allocator.free(bytes);
return error.InvalidUtf8;
}
return bytes;
}
test "encode matches the known vectors" {
const t = std.testing;
const a = t.allocator;
const hello = try b64Encode(a, "Hello, world!");
defer a.free(hello);
try t.expectEqualStrings("SGVsbG8sIHdvcmxkIQ==", hello);
const one = try b64Encode(a, "a");
defer a.free(one);
try t.expectEqualStrings("YQ==", one);
const three = try b64Encode(a, "abc");
defer a.free(three);
try t.expectEqualStrings("YWJj", three);
}
test "decode strips whitespace and round-trips UTF-8" {
const t = std.testing;
const a = t.allocator;
const text = try b64Decode(a, "aGVs\nbG8g d29ybGQ=");
defer a.free(text);
try t.expectEqualStrings("hello world", text);
const unicode = "héllo 🌍";
const encoded = try b64Encode(a, unicode);
defer a.free(encoded);
const decoded = try b64Decode(a, encoded);
defer a.free(decoded);
try t.expectEqualStrings(unicode, decoded);
}
test "decode rejects malformed input" {
const t = std.testing;
const a = t.allocator;
try t.expectError(error.BadLength, b64Decode(a, "SGVsbG8"));
try t.expectError(error.IllegalCharacter, b64Decode(a, "SGVsbG8*"));
// "/w==" decodes to the single byte 0xFF, which is not valid UTF-8.
try t.expectError(error.InvalidUtf8, b64Decode(a, "/w=="));
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →