Skip to content

Base64 Encode / Decode — C++ source

Encode text to Base64 or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// base64 — UTF-8 safe Base64 encode/decode.
//
// Language: C++ (C++17, standard library only — C++ has no Base64 in its
//           standard library, so the codec is hand-rolled like rust.rs)
// Source:   CosmoDev polyglot showcase port of the `base64` tool, ported
//           from src/lib/base64.ts (the canonical TypeScript implementation);
//           algorithm and structure mirror src/tool-sources/base64/rust.rs.
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// std::string is a byte container, so a string holding UTF-8 text is already
// the byte sequence Base64 operates on — there is no separate "encode to
// UTF-8" step, unlike the TypeScript port's TextEncoder. On decode the
// output bytes are validated against RFC 3629 (the TextDecoder step), and
// malformed input throws std::invalid_argument — the C++ reading of the TS
// port's "throw on invalid input" contract.
//
// Build: c++ -std=c++17 cpp.cpp && ./a.out

#include <array>
#include <cstdint>
#include <cstdio>
#include <stdexcept>
#include <string>

namespace {

/** Standard Base64 alphabet (RFC 4648). The position of each byte in this
 *  table is its 6-bit value — the same alphabet btoa emits in the browser. */
constexpr char kAlphabet[] =
    "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";

/** Sentinel for "this byte is not part of the Base64 alphabet". */
constexpr int8_t kInvalid = -1;
/** Sentinel for "this byte is the '=' padding character". */
constexpr int8_t kPadding = -2;

/** Build a 256-entry lookup table mapping an ASCII byte to its 6-bit value,
 *  or one of the sentinels. Filled once (thread-safe static init) and reused
 *  for every decode. */
const std::array<int8_t, 256>& decode_table() {
    static const std::array<int8_t, 256> table = [] {
        std::array<int8_t, 256> filled{};
        filled.fill(kInvalid);
        for (size_t value = 0; value < 64; ++value) {
            filled[static_cast<unsigned char>(kAlphabet[value])] =
                static_cast<int8_t>(value);
        }
        filled[static_cast<unsigned char>('=')] = kPadding;
        return filled;
    }();
    return table;
}

/** True when c is one of the six whitespace characters JavaScript's \s
 *  recognizes: space, \t, \n, \v, \f, \r. */
bool is_space(char c) {
    return c == ' ' || c == '\t' || c == '\n' || c == '\v' || c == '\f' ||
           c == '\r';
}

/** Validate a byte range as UTF-8 per RFC 3629: rejects truncated sequences,
 *  overlong encodings, UTF-16 surrogate halves (U+D800..U+DFFF) and code
 *  points above U+10FFFF. This is the TextDecoder step of the TS port. */
bool utf8_valid(const std::string& bytes) {
    const auto* p = reinterpret_cast<const unsigned char*>(bytes.data());
    const size_t n = bytes.size();
    size_t i = 0;

    while (i < n) {
        unsigned char b = p[i];
        unsigned cp = 0;

        if (b < 0x80) {                      // 0xxxxxxx: ASCII
            i += 1;
        } else if ((b & 0xE0) == 0xC0) {     // 110xxxxx: 2-byte lead
            if (b < 0xC2) return false;      // C0/C1 would be overlong
            if (i + 1 >= n || (p[i + 1] & 0xC0) != 0x80) return false;
            i += 2;
        } else if ((b & 0xF0) == 0xE0) {     // 1110xxxx: 3-byte lead
            if (i + 2 >= n || (p[i + 1] & 0xC0) != 0x80 ||
                (p[i + 2] & 0xC0) != 0x80) return false;
            cp = static_cast<unsigned>(b & 0x0F) << 12 |
                 static_cast<unsigned>(p[i + 1] & 0x3F) << 6 |
                 static_cast<unsigned>(p[i + 2] & 0x3F);
            if (cp < 0x800) return false;                    // overlong
            if (cp >= 0xD800 && cp <= 0xDFFF) return false;  // surrogate
            i += 3;
        } else if ((b & 0xF8) == 0xF0) {     // 11110xxx: 4-byte lead
            if (i + 3 >= n || (p[i + 1] & 0xC0) != 0x80 ||
                (p[i + 2] & 0xC0) != 0x80 || (p[i + 3] & 0xC0) != 0x80) {
                return false;
            }
            cp = static_cast<unsigned>(b & 0x07) << 18 |
                 static_cast<unsigned>(p[i + 1] & 0x3F) << 12 |
                 static_cast<unsigned>(p[i + 2] & 0x3F) << 6 |
                 static_cast<unsigned>(p[i + 3] & 0x3F);
            if (cp < 0x10000 || cp > 0x10FFFF) return false;
            i += 4;
        } else {
            return false;  // stray continuation byte or invalid lead byte
        }
    }
    return true;
}

}  // namespace

/**
 * Encode a UTF-8 string into standard, padded Base64.
 *
 * Walks the bytes in 3-byte groups, emitting four 6-bit indices per group.
 * A trailing partial group (1 or 2 bytes) is padded with '=' so the output
 * length is always a multiple of 4 — the same shape as btoa in the browser.
 */
std::string b64encode(const std::string& input) {
    const auto* bytes = reinterpret_cast<const unsigned char*>(input.data());
    std::string out;
    out.reserve((input.size() + 2) / 3 * 4);

    size_t i = 0;
    while (i + 3 <= input.size()) {
        unsigned triple = (static_cast<unsigned>(bytes[i]) << 16) |
                          (static_cast<unsigned>(bytes[i + 1]) << 8) |
                          static_cast<unsigned>(bytes[i + 2]);
        out += kAlphabet[(triple >> 18) & 0x3F];
        out += kAlphabet[(triple >> 12) & 0x3F];
        out += kAlphabet[(triple >> 6) & 0x3F];
        out += kAlphabet[triple & 0x3F];
        i += 3;
    }

    // Trailing 1 or 2 bytes, padded so the output stays a multiple of 4.
    const size_t remainder = input.size() - i;
    if (remainder == 1) {
        const unsigned triple = static_cast<unsigned>(bytes[i]) << 16;
        out += kAlphabet[(triple >> 18) & 0x3F];
        out += kAlphabet[(triple >> 12) & 0x3F];
        out += "==";
    } else if (remainder == 2) {
        const unsigned triple = (static_cast<unsigned>(bytes[i]) << 16) |
                                (static_cast<unsigned>(bytes[i + 1]) << 8);
        out += kAlphabet[(triple >> 18) & 0x3F];
        out += kAlphabet[(triple >> 12) & 0x3F];
        out += kAlphabet[(triple >> 6) & 0x3F];
        out += '=';
    }

    return out;
}

/**
 * Decode standard Base64 back into the original UTF-8 text.
 *
 * Whitespace inside the input is stripped first (the six characters \s
 * matches), so line-wrapped Base64 decodes cleanly. Any malformed input —
 * an illegal character, a length that is not a multiple of 4, or decoded
 * bytes that are not valid UTF-8 — throws std::invalid_argument, matching
 * the TS port's "throw on invalid input" contract.
 */
std::string b64decode(const std::string& input) {
    const auto& table = decode_table();

    // Drop every whitespace character, keeping the surviving ASCII bytes.
    std::string cleaned;
    cleaned.reserve(input.size());
    for (char c : input) {
        if (!is_space(c)) cleaned += c;
    }

    // Standard Base64 with padding is always a multiple of 4 characters.
    if (cleaned.size() % 4 != 0) {
        throw std::invalid_argument(
            "invalid base64: length is not a multiple of 4");
    }

    // Count trailing '=' padding (0, 1, or 2 in well-formed input).
    size_t padding = 0;
    while (padding < 2 && cleaned.size() > padding &&
           cleaned[cleaned.size() - 1 - padding] == '=') {
        ++padding;
    }

    std::string bytes;
    bytes.reserve(cleaned.size() * 3 / 4);

    const size_t n = cleaned.size();
    for (size_t i = 0; i < n; i += 4) {
        // Read four sextets, validating each against the lookup table.
        unsigned char sextets[4] = {0, 0, 0, 0};
        for (size_t j = 0; j < 4; ++j) {
            const int8_t entry =
                table[static_cast<unsigned char>(cleaned[i + j])];
            if (entry == kInvalid) {
                throw std::invalid_argument(
                    "invalid base64: illegal character at byte " +
                    std::to_string(i + j));
            }
            // Padding contributes zero bits.
            sextets[j] = static_cast<unsigned char>(entry == kPadding ? 0
                                                                      : entry);
        }

        const unsigned triple = (static_cast<unsigned>(sextets[0]) << 18) |
                                (static_cast<unsigned>(sextets[1]) << 12) |
                                (static_cast<unsigned>(sextets[2]) << 6) |
                                static_cast<unsigned>(sextets[3]);
        const bool last_group = (i + 4 == n);

        bytes += static_cast<char>((triple >> 16) & 0xFF);  // byte 0: always
        if (!(last_group && padding == 2)) {
            bytes += static_cast<char>((triple >> 8) & 0xFF);  // byte 1
        }
        if (!(last_group && padding >= 1)) {
            bytes += static_cast<char>(triple & 0xFF);  // byte 2
        }
    }

    // Re-interpret the decoded bytes as UTF-8 (the TextDecoder step).
    if (!utf8_valid(bytes)) {
        throw std::invalid_argument(
            "invalid base64: decoded bytes are not valid UTF-8");
    }
    return bytes;
}

/* -------------------------------------------------------------------------
 * Demo
 * ------------------------------------------------------------------------- */

int main() {
    std::printf("encode: %s\n", b64encode("Hello, world!").c_str());

    try {
        std::printf("decode (whitespace stripped): %s\n",
                    b64decode("aGVs\nbG8g d29ybGQ=").c_str());

        // Multi-byte UTF-8 survives the round trip.
        const std::string unicode = "héllo 🌍";
        const std::string round_trip = b64decode(b64encode(unicode));
        std::printf("round trip: %s\n",
                    round_trip == unicode ? "ok" : round_trip.c_str());
    } catch (const std::invalid_argument& e) {
        std::printf("unexpected failure: %s\n", e.what());
    }

    try {
        b64decode("SGVsbG8*");
    } catch (const std::invalid_argument& e) {
        std::printf("illegal character: %s\n", e.what());
    }

    // "/w==" decodes to the single byte 0xFF, which is not valid UTF-8.
    try {
        b64decode("/w==");
    } catch (const std::invalid_argument& e) {
        std::printf("invalid UTF-8: %s\n", e.what());
    }

    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →