Skip to content

Base64 Encode / Decode — C source

Encode text to Base64 or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/*
 * base64 — UTF-8 safe Base64 encode/decode.
 *
 * Language: C (C11, standard library only — C has no Base64 in its standard
 *           library, so the codec is hand-rolled like rust.rs)
 * Source:   CosmoDev polyglot showcase port of the `base64` tool, ported
 *           from src/lib/base64.ts (the canonical TypeScript implementation);
 *           algorithm and structure mirror src/tool-sources/base64/rust.rs.
 * License:  display source — part of CosmoDev's polyglot tool pages.
 *
 * C strings are byte sequences, so a `char *` holding UTF-8 text is already
 * the byte sequence Base64 operates on — there is no separate "encode to
 * UTF-8" step, unlike the TypeScript port's TextEncoder. On decode the
 * output bytes are validated against RFC 3629 before being returned; that
 * validation is the role TextDecoder plays in the TypeScript original.
 *
 * Errors follow the TS port's "throw on invalid input" contract as closely
 * as C allows: b64_decode returns NULL and writes a reason into the caller's
 * error buffer.
 *
 * Build: cc -std=c11 c.c && ./a.out
 */

#include <ctype.h>
#include <stdarg.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

/** Standard Base64 alphabet (RFC 4648). The position of each byte in this
 *  table is its 6-bit value — the same alphabet btoa emits in the browser. */
static const char B64_ALPHABET[] =
    "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";

/** b64_value() sentinel: the byte is not part of the Base64 alphabet. */
#define B64_INVALID (-1)
/** b64_value() sentinel: the byte is the '=' padding character. */
#define B64_PADDING (-2)

/** Write a formatted failure reason into the caller's error buffer. */
static void b64_fail(char *error, size_t error_size, const char *fmt, ...)
{
    if (error == NULL || error_size == 0) return;

    va_list args;
    va_start(args, fmt);
    vsnprintf(error, error_size, fmt, args);
    va_end(args);
}

/** Map an ASCII byte to its 6-bit value, or a B64_* sentinel. A memchr scan
 *  over the 64-byte alphabet stands in for the 256-entry lookup table
 *  rust.rs builds — same semantics, less machinery. */
static int b64_value(unsigned char byte)
{
    if (byte == '=') return B64_PADDING;

    const char *hit = memchr(B64_ALPHABET, byte, 64);
    return (hit != NULL) ? (int)(hit - B64_ALPHABET) : B64_INVALID;
}

/** True when byte is one of the six whitespace characters JavaScript's \s
 *  and C's isspace() both recognize: space, \t, \n, \v, \f, \r. */
static bool b64_is_space(unsigned char byte)
{
    return byte == ' ' || byte == '\t' || byte == '\n'
        || byte == '\v' || byte == '\f' || byte == '\r';
}

/**
 * Validate a byte range as UTF-8 per RFC 3629: rejects truncated sequences,
 * overlong encodings, UTF-16 surrogate halves (U+D800..U+DFFF) and code
 * points above U+10FFFF. This is the TextDecoder step of the TS port.
 */
static bool b64_utf8_valid(const unsigned char *bytes, size_t len)
{
    size_t i = 0;

    while (i < len) {
        unsigned char b = bytes[i];
        unsigned cp;

        if (b < 0x80) {                       /* 0xxxxxxx: ASCII */
            i += 1;
            continue;
        }
        if ((b & 0xE0) == 0xC0) {             /* 110xxxxx: 2-byte lead */
            if (b < 0xC2) return false;       /* C0/C1 would be overlong */
            if (i + 1 >= len
                || (bytes[i + 1] & 0xC0) != 0x80) return false;
            i += 2;
            continue;
        }
        if ((b & 0xF0) == 0xE0) {             /* 1110xxxx: 3-byte lead */
            if (i + 2 >= len
                || (bytes[i + 1] & 0xC0) != 0x80
                || (bytes[i + 2] & 0xC0) != 0x80) return false;
            cp = ((unsigned)(b & 0x0F) << 12)
               | ((unsigned)(bytes[i + 1] & 0x3F) << 6)
               | (unsigned)(bytes[i + 2] & 0x3F);
            if (cp < 0x800) return false;                    /* overlong */
            if (cp >= 0xD800 && cp <= 0xDFFF) return false;  /* surrogate */
            i += 3;
            continue;
        }
        if ((b & 0xF8) == 0xF0) {             /* 11110xxx: 4-byte lead */
            if (i + 3 >= len
                || (bytes[i + 1] & 0xC0) != 0x80
                || (bytes[i + 2] & 0xC0) != 0x80
                || (bytes[i + 3] & 0xC0) != 0x80) return false;
            cp = ((unsigned)(b & 0x07) << 18)
               | ((unsigned)(bytes[i + 1] & 0x3F) << 12)
               | ((unsigned)(bytes[i + 2] & 0x3F) << 6)
               | (unsigned)(bytes[i + 3] & 0x3F);
            if (cp < 0x10000 || cp > 0x10FFFF) return false;
            i += 4;
            continue;
        }
        return false;   /* stray continuation byte or invalid lead byte */
    }
    return true;
}

/**
 * Encode a NUL-terminated UTF-8 string into standard, padded Base64.
 *
 * Walks the bytes in 3-byte groups, emitting four 6-bit indices per group.
 * A trailing partial group (1 or 2 bytes) is padded with '=' so the output
 * length is always a multiple of 4 — the same shape as btoa in the browser.
 *
 * Returns a malloc'd NUL-terminated string, or NULL on allocation failure.
 */
char *b64_encode(const char *text)
{
    const unsigned char *bytes = (const unsigned char *)text;
    size_t len = strlen(text);
    char *out = malloc(((len + 2) / 3) * 4 + 1);

    if (out == NULL) return NULL;

    size_t i = 0, o = 0;

    /* Complete 3-byte chunks -> four Base64 characters. */
    while (i + 3 <= len) {
        unsigned triple = ((unsigned)bytes[i] << 16)
                        | ((unsigned)bytes[i + 1] << 8)
                        | (unsigned)bytes[i + 2];
        out[o++] = B64_ALPHABET[(triple >> 18) & 0x3F];
        out[o++] = B64_ALPHABET[(triple >> 12) & 0x3F];
        out[o++] = B64_ALPHABET[(triple >> 6) & 0x3F];
        out[o++] = B64_ALPHABET[triple & 0x3F];
        i += 3;
    }

    /* Trailing 1 or 2 bytes, padded so the output stays a multiple of 4. */
    switch (len - i) {
    case 1: {
        unsigned triple = (unsigned)bytes[i] << 16;
        out[o++] = B64_ALPHABET[(triple >> 18) & 0x3F];
        out[o++] = B64_ALPHABET[(triple >> 12) & 0x3F];
        out[o++] = '=';
        out[o++] = '=';
        break;
    }
    case 2: {
        unsigned triple = ((unsigned)bytes[i] << 16)
                        | ((unsigned)bytes[i + 1] << 8);
        out[o++] = B64_ALPHABET[(triple >> 18) & 0x3F];
        out[o++] = B64_ALPHABET[(triple >> 12) & 0x3F];
        out[o++] = B64_ALPHABET[(triple >> 6) & 0x3F];
        out[o++] = '=';
        break;
    }
    default:
        break;  /* no remainder: nothing left to emit */
    }

    out[o] = '\0';
    return out;
}

/**
 * Decode a standard Base64 string back into the original NUL-terminated
 * UTF-8 text.
 *
 * Whitespace inside the input is stripped first (the six characters \s
 * matches), so line-wrapped Base64 decodes cleanly. Any malformed input —
 * an illegal character, a length that is not a multiple of 4, or decoded
 * bytes that are not valid UTF-8 — returns NULL with a reason written into
 * `error` (if non-NULL), matching the TS port's "throw on invalid input"
 * contract as closely as C allows.
 */
char *b64_decode(const char *input, char *error, size_t error_size)
{
    size_t len = strlen(input);
    unsigned char *cleaned = malloc(len + 1);

    if (cleaned == NULL) {
        b64_fail(error, error_size, "out of memory");
        return NULL;
    }

    /* Drop every whitespace byte, keeping the surviving ASCII bytes. */
    size_t n = 0;
    for (size_t i = 0; i < len; i++) {
        if (!b64_is_space((unsigned char)input[i])) {
            cleaned[n++] = (unsigned char)input[i];
        }
    }

    /* Standard Base64 with padding is always a multiple of 4 characters. */
    if (n % 4 != 0) {
        b64_fail(error, error_size,
                 "invalid base64: length is not a multiple of 4");
        free(cleaned);
        return NULL;
    }

    /* Count trailing '=' padding (0, 1, or 2 in well-formed input). */
    size_t padding = 0;
    while (padding < 2 && n > padding && cleaned[n - 1 - padding] == '=') {
        padding++;
    }

    unsigned char *bytes = malloc(n / 4 * 3 + 1);
    if (bytes == NULL) {
        b64_fail(error, error_size, "out of memory");
        free(cleaned);
        return NULL;
    }

    size_t o = 0;
    for (size_t i = 0; i < n; i += 4) {
        /* Read four sextets, validating each against the alphabet. */
        unsigned char sextets[4] = {0, 0, 0, 0};
        for (size_t j = 0; j < 4; j++) {
            int value = b64_value(cleaned[i + j]);
            if (value == B64_INVALID) {
                b64_fail(error, error_size,
                         "invalid base64: illegal character at byte %zu",
                         i + j);
                free(cleaned);
                free(bytes);
                return NULL;
            }
            /* Padding contributes zero bits. */
            sextets[j] = (unsigned char)(value == B64_PADDING ? 0 : value);
        }

        unsigned triple = ((unsigned)sextets[0] << 18)
                        | ((unsigned)sextets[1] << 12)
                        | ((unsigned)sextets[2] << 6)
                        | (unsigned)sextets[3];
        bool last_group = (i + 4 == n);

        bytes[o++] = (unsigned char)((triple >> 16) & 0xFF);  /* byte 0: always present */
        if (!(last_group && padding == 2)) {
            bytes[o++] = (unsigned char)((triple >> 8) & 0xFF);  /* byte 1 */
        }
        if (!(last_group && padding >= 1)) {
            bytes[o++] = (unsigned char)(triple & 0xFF);         /* byte 2 */
        }
    }
    free(cleaned);

    /* Re-interpret the decoded bytes as UTF-8 before handing them back. */
    if (!b64_utf8_valid(bytes, o)) {
        b64_fail(error, error_size,
                 "invalid base64: decoded bytes are not valid UTF-8");
        free(bytes);
        return NULL;
    }

    bytes[o] = '\0';
    return (char *)bytes;
}

/* -------------------------------------------------------------------------
 * Demo
 * ------------------------------------------------------------------------- */

int main(void)
{
    char error[128];

    char *encoded = b64_encode("Hello, world!");
    printf("encode: %s\n", encoded ? encoded : "(allocation failed)");
    free(encoded);

    char *decoded = b64_decode("aGVs\nbG8g d29ybGQ=", error, sizeof error);
    printf("decode (whitespace stripped): %s\n", decoded ? decoded : error);
    free(decoded);

    /* Multi-byte UTF-8 survives the round trip: "héllo 🌍" as raw bytes. */
    const char *unicode = "h\xc3\xa9llo \xf0\x9f\x8c\x8d";
    char *unicode_b64 = b64_encode(unicode);
    char *round_trip = unicode_b64 ? b64_decode(unicode_b64, error, sizeof error)
                                   : NULL;
    printf("round trip: %s\n",
           (round_trip != NULL && strcmp(round_trip, unicode) == 0)
               ? "ok" : (round_trip != NULL ? round_trip : error));
    free(round_trip);
    free(unicode_b64);

    decoded = b64_decode("SGVsbG8*", error, sizeof error);
    printf("illegal character: %s\n", decoded ? decoded : error);
    free(decoded);

    /* "/w==" decodes to the single byte 0xFF, which is not valid UTF-8. */
    decoded = b64_decode("/w==", error, sizeof error);
    printf("invalid UTF-8: %s\n", decoded ? decoded : error);
    free(decoded);

    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →