Base64 Encode / Decode — C++ source
Encode text to Base64 or decode it back. UTF-8 safe, runs entirely in your browser, with a shareable link to your exact input.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// base64 — UTF-8 safe Base64 encode/decode.
//
// Language: C++ (C++17, standard library only — C++ has no Base64 in its
// standard library, so the codec is hand-rolled like rust.rs)
// Source: CosmoDev polyglot showcase port of the `base64` tool, ported
// from src/lib/base64.ts (the canonical TypeScript implementation);
// algorithm and structure mirror src/tool-sources/base64/rust.rs.
// License: display source — part of CosmoDev's polyglot tool pages.
//
// std::string is a byte container, so a string holding UTF-8 text is already
// the byte sequence Base64 operates on — there is no separate "encode to
// UTF-8" step, unlike the TypeScript port's TextEncoder. On decode the
// output bytes are validated against RFC 3629 (the TextDecoder step), and
// malformed input throws std::invalid_argument — the C++ reading of the TS
// port's "throw on invalid input" contract.
//
// Build: c++ -std=c++17 cpp.cpp && ./a.out
#include <array>
#include <cstdint>
#include <cstdio>
#include <stdexcept>
#include <string>
namespace {
/** Standard Base64 alphabet (RFC 4648). The position of each byte in this
* table is its 6-bit value — the same alphabet btoa emits in the browser. */
constexpr char kAlphabet[] =
"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";
/** Sentinel for "this byte is not part of the Base64 alphabet". */
constexpr int8_t kInvalid = -1;
/** Sentinel for "this byte is the '=' padding character". */
constexpr int8_t kPadding = -2;
/** Build a 256-entry lookup table mapping an ASCII byte to its 6-bit value,
* or one of the sentinels. Filled once (thread-safe static init) and reused
* for every decode. */
const std::array<int8_t, 256>& decode_table() {
static const std::array<int8_t, 256> table = [] {
std::array<int8_t, 256> filled{};
filled.fill(kInvalid);
for (size_t value = 0; value < 64; ++value) {
filled[static_cast<unsigned char>(kAlphabet[value])] =
static_cast<int8_t>(value);
}
filled[static_cast<unsigned char>('=')] = kPadding;
return filled;
}();
return table;
}
/** True when c is one of the six whitespace characters JavaScript's \s
* recognizes: space, \t, \n, \v, \f, \r. */
bool is_space(char c) {
return c == ' ' || c == '\t' || c == '\n' || c == '\v' || c == '\f' ||
c == '\r';
}
/** Validate a byte range as UTF-8 per RFC 3629: rejects truncated sequences,
* overlong encodings, UTF-16 surrogate halves (U+D800..U+DFFF) and code
* points above U+10FFFF. This is the TextDecoder step of the TS port. */
bool utf8_valid(const std::string& bytes) {
const auto* p = reinterpret_cast<const unsigned char*>(bytes.data());
const size_t n = bytes.size();
size_t i = 0;
while (i < n) {
unsigned char b = p[i];
unsigned cp = 0;
if (b < 0x80) { // 0xxxxxxx: ASCII
i += 1;
} else if ((b & 0xE0) == 0xC0) { // 110xxxxx: 2-byte lead
if (b < 0xC2) return false; // C0/C1 would be overlong
if (i + 1 >= n || (p[i + 1] & 0xC0) != 0x80) return false;
i += 2;
} else if ((b & 0xF0) == 0xE0) { // 1110xxxx: 3-byte lead
if (i + 2 >= n || (p[i + 1] & 0xC0) != 0x80 ||
(p[i + 2] & 0xC0) != 0x80) return false;
cp = static_cast<unsigned>(b & 0x0F) << 12 |
static_cast<unsigned>(p[i + 1] & 0x3F) << 6 |
static_cast<unsigned>(p[i + 2] & 0x3F);
if (cp < 0x800) return false; // overlong
if (cp >= 0xD800 && cp <= 0xDFFF) return false; // surrogate
i += 3;
} else if ((b & 0xF8) == 0xF0) { // 11110xxx: 4-byte lead
if (i + 3 >= n || (p[i + 1] & 0xC0) != 0x80 ||
(p[i + 2] & 0xC0) != 0x80 || (p[i + 3] & 0xC0) != 0x80) {
return false;
}
cp = static_cast<unsigned>(b & 0x07) << 18 |
static_cast<unsigned>(p[i + 1] & 0x3F) << 12 |
static_cast<unsigned>(p[i + 2] & 0x3F) << 6 |
static_cast<unsigned>(p[i + 3] & 0x3F);
if (cp < 0x10000 || cp > 0x10FFFF) return false;
i += 4;
} else {
return false; // stray continuation byte or invalid lead byte
}
}
return true;
}
} // namespace
/**
* Encode a UTF-8 string into standard, padded Base64.
*
* Walks the bytes in 3-byte groups, emitting four 6-bit indices per group.
* A trailing partial group (1 or 2 bytes) is padded with '=' so the output
* length is always a multiple of 4 — the same shape as btoa in the browser.
*/
std::string b64encode(const std::string& input) {
const auto* bytes = reinterpret_cast<const unsigned char*>(input.data());
std::string out;
out.reserve((input.size() + 2) / 3 * 4);
size_t i = 0;
while (i + 3 <= input.size()) {
unsigned triple = (static_cast<unsigned>(bytes[i]) << 16) |
(static_cast<unsigned>(bytes[i + 1]) << 8) |
static_cast<unsigned>(bytes[i + 2]);
out += kAlphabet[(triple >> 18) & 0x3F];
out += kAlphabet[(triple >> 12) & 0x3F];
out += kAlphabet[(triple >> 6) & 0x3F];
out += kAlphabet[triple & 0x3F];
i += 3;
}
// Trailing 1 or 2 bytes, padded so the output stays a multiple of 4.
const size_t remainder = input.size() - i;
if (remainder == 1) {
const unsigned triple = static_cast<unsigned>(bytes[i]) << 16;
out += kAlphabet[(triple >> 18) & 0x3F];
out += kAlphabet[(triple >> 12) & 0x3F];
out += "==";
} else if (remainder == 2) {
const unsigned triple = (static_cast<unsigned>(bytes[i]) << 16) |
(static_cast<unsigned>(bytes[i + 1]) << 8);
out += kAlphabet[(triple >> 18) & 0x3F];
out += kAlphabet[(triple >> 12) & 0x3F];
out += kAlphabet[(triple >> 6) & 0x3F];
out += '=';
}
return out;
}
/**
* Decode standard Base64 back into the original UTF-8 text.
*
* Whitespace inside the input is stripped first (the six characters \s
* matches), so line-wrapped Base64 decodes cleanly. Any malformed input —
* an illegal character, a length that is not a multiple of 4, or decoded
* bytes that are not valid UTF-8 — throws std::invalid_argument, matching
* the TS port's "throw on invalid input" contract.
*/
std::string b64decode(const std::string& input) {
const auto& table = decode_table();
// Drop every whitespace character, keeping the surviving ASCII bytes.
std::string cleaned;
cleaned.reserve(input.size());
for (char c : input) {
if (!is_space(c)) cleaned += c;
}
// Standard Base64 with padding is always a multiple of 4 characters.
if (cleaned.size() % 4 != 0) {
throw std::invalid_argument(
"invalid base64: length is not a multiple of 4");
}
// Count trailing '=' padding (0, 1, or 2 in well-formed input).
size_t padding = 0;
while (padding < 2 && cleaned.size() > padding &&
cleaned[cleaned.size() - 1 - padding] == '=') {
++padding;
}
std::string bytes;
bytes.reserve(cleaned.size() * 3 / 4);
const size_t n = cleaned.size();
for (size_t i = 0; i < n; i += 4) {
// Read four sextets, validating each against the lookup table.
unsigned char sextets[4] = {0, 0, 0, 0};
for (size_t j = 0; j < 4; ++j) {
const int8_t entry =
table[static_cast<unsigned char>(cleaned[i + j])];
if (entry == kInvalid) {
throw std::invalid_argument(
"invalid base64: illegal character at byte " +
std::to_string(i + j));
}
// Padding contributes zero bits.
sextets[j] = static_cast<unsigned char>(entry == kPadding ? 0
: entry);
}
const unsigned triple = (static_cast<unsigned>(sextets[0]) << 18) |
(static_cast<unsigned>(sextets[1]) << 12) |
(static_cast<unsigned>(sextets[2]) << 6) |
static_cast<unsigned>(sextets[3]);
const bool last_group = (i + 4 == n);
bytes += static_cast<char>((triple >> 16) & 0xFF); // byte 0: always
if (!(last_group && padding == 2)) {
bytes += static_cast<char>((triple >> 8) & 0xFF); // byte 1
}
if (!(last_group && padding >= 1)) {
bytes += static_cast<char>(triple & 0xFF); // byte 2
}
}
// Re-interpret the decoded bytes as UTF-8 (the TextDecoder step).
if (!utf8_valid(bytes)) {
throw std::invalid_argument(
"invalid base64: decoded bytes are not valid UTF-8");
}
return bytes;
}
/* -------------------------------------------------------------------------
* Demo
* ------------------------------------------------------------------------- */
int main() {
std::printf("encode: %s\n", b64encode("Hello, world!").c_str());
try {
std::printf("decode (whitespace stripped): %s\n",
b64decode("aGVs\nbG8g d29ybGQ=").c_str());
// Multi-byte UTF-8 survives the round trip.
const std::string unicode = "héllo 🌍";
const std::string round_trip = b64decode(b64encode(unicode));
std::printf("round trip: %s\n",
round_trip == unicode ? "ok" : round_trip.c_str());
} catch (const std::invalid_argument& e) {
std::printf("unexpected failure: %s\n", e.what());
}
try {
b64decode("SGVsbG8*");
} catch (const std::invalid_argument& e) {
std::printf("illegal character: %s\n", e.what());
}
// "/w==" decodes to the single byte 0xFF, which is not valid UTF-8.
try {
b64decode("/w==");
} catch (const std::invalid_argument& e) {
std::printf("invalid UTF-8: %s\n", e.what());
}
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →