JSON ↔ CSV Converter — C++ source
Convert a JSON array of objects to CSV and back. Handles quoted fields, embedded commas, newlines and escaped quotes (RFC 4180). 100% in-browser.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// =============================================================================
// json-csv — C++ port
// =============================================================================
// Convert between JSON and RFC 4180 CSV in either direction:
// • json_to_csv — serialize a JSON document (object or array of objects) to CSV
// • csv_to_json — parse RFC 4180 CSV (with quoting) into a vector of row objects
//
// Language: C++17 — standard library only; minimal embedded JSON parser.
// Source: CosmoDev polyglot showcase port of json-csv,
// ported from src/lib/csv.ts (the canonical, live TypeScript lib).
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Pure and deterministic — depends only on its inputs. RFC 4180 quoting: any
// field containing a comma, double quote, carriage return, or line feed is
// wrapped in double quotes, and each embedded quote is doubled.
//
// This is display source — part of CosmoDev's polyglot tool pages.
// =============================================================================
// C++'s stdlib has no JSON support, so this file is self-contained: a small
// recursive-descent parser, mirroring the Rust sibling snippet (which would
// use an ecosystem crate such as `serde_json` instead). Object pairs live in
// a vector so key order — which is observable (it determines CSV column
// order) — is preserved, exactly as in the canonical lib.
#include <cctype>
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <optional>
#include <stdexcept>
#include <string>
#include <unordered_set>
#include <utility>
#include <vector>
namespace jsoncsv {
// ---------------------------------------------------------------------------
// JSON value tree
// ---------------------------------------------------------------------------
struct Value {
enum class Kind { Null, Bool, Number, Str, Array, Object };
Kind kind = Kind::Null;
bool b = false;
double num = 0.0;
std::string str;
std::vector<Value> items; // Array
std::vector<std::pair<std::string, Value>> pairs; // Object, insertion-ordered
static Value null() { return Value{}; }
static Value boolean(bool v) { Value x; x.kind = Kind::Bool; x.b = v; return x; }
static Value number(double v) { Value x; x.kind = Kind::Number; x.num = v; return x; }
static Value string(std::string v) { Value x; x.kind = Kind::Str; x.str = std::move(v); return x; }
static Value array(std::vector<Value> v) { Value x; x.kind = Kind::Array; x.items = std::move(v); return x; }
static Value object(std::vector<std::pair<std::string, Value>> v) { Value x; x.kind = Kind::Object; x.pairs = std::move(v); return x; }
};
// ---------------------------------------------------------------------------
// Minimal JSON parser
// ---------------------------------------------------------------------------
// Compact recursive-descent parser. Sufficient for any RFC 8259 document a
// caller is likely to feed this tool.
class ParseError : public std::runtime_error {
public:
explicit ParseError(const std::string& msg) : std::runtime_error(msg) {}
};
class Parser {
public:
explicit Parser(const std::string& input) : input_(input) {}
Value parse_value() {
skip_ws();
require_more("unexpected end of input");
const char c = peek();
switch (c) {
case '{': return parse_object();
case '[': return parse_array();
case '"': return Value::string(parse_string());
case 't': case 'f': return parse_bool();
case 'n': return parse_null();
default:
if (c == '-' || (c >= '0' && c <= '9')) return parse_number();
throw ParseError(std::string("unexpected character '") + c + "'");
}
}
private:
const std::string& input_;
size_t pos_ = 0;
void skip_ws() {
while (pos_ < input_.size()) {
switch (input_[pos_]) {
case ' ': case '\t': case '\n': case '\r': pos_++; break;
default: return;
}
}
}
void require_more(const char* msg) const {
if (pos_ >= input_.size()) throw ParseError(msg);
}
char peek() const { return input_[pos_]; }
Value parse_object() {
pos_++; // {
std::vector<std::pair<std::string, Value>> pairs;
skip_ws();
require_more("expected key or '}' in object");
if (peek() == '}') {
pos_++;
return Value::object(std::move(pairs));
}
while (true) {
skip_ws();
if (peek() != '"') throw ParseError("expected string key in object");
std::string key = parse_string();
skip_ws();
if (peek() != ':') throw ParseError("expected ':' after object key");
pos_++;
Value val = parse_value();
// First occurrence of a key wins, matching JS object-literal semantics.
bool dup = false;
for (const auto& p : pairs) {
if (p.first == key) { dup = true; break; }
}
if (!dup) pairs.emplace_back(std::move(key), std::move(val));
skip_ws();
switch (peek()) {
case ',': pos_++; break;
case '}': pos_++; return Value::object(std::move(pairs));
default: throw ParseError("expected ',' or '}' in object");
}
}
}
Value parse_array() {
pos_++; // [
std::vector<Value> items;
skip_ws();
require_more("expected value or ']' in array");
if (peek() == ']') {
pos_++;
return Value::array(std::move(items));
}
while (true) {
items.push_back(parse_value());
skip_ws();
switch (peek()) {
case ',': pos_++; break;
case ']': pos_++; return Value::array(std::move(items));
default: throw ParseError("expected ',' or ']' in array");
}
}
}
std::string parse_string() {
pos_++; // opening quote
std::string out;
while (pos_ < input_.size()) {
const char c = input_[pos_++];
if (c == '"') return out;
if (c == '\\') {
require_more("trailing escape");
const char e = input_[pos_++];
switch (e) {
case '"': out += '"'; break;
case '\\': out += '\\'; break;
case '/': out += '/'; break;
case 'n': out += '\n'; break;
case 't': out += '\t'; break;
case 'r': out += '\r'; break;
case 'b': out += '\b'; break;
case 'f': out += '\f'; break;
case 'u': append_codepoint(out, parse_codepoint()); break;
default: throw ParseError(std::string("bad escape \\") + e);
}
} else {
out += c;
}
}
throw ParseError("unterminated string");
}
// Encode a codepoint as UTF-8 (JSON strings decode to UTF-8 in every
// sibling port; \u escapes outside the BMP arrive as surrogate pairs).
static void append_codepoint(std::string& out, uint32_t cp) {
if (cp < 0x80) {
out += static_cast<char>(cp);
} else if (cp < 0x800) {
out += static_cast<char>(0xC0 | (cp >> 6));
out += static_cast<char>(0x80 | (cp & 0x3F));
} else if (cp < 0x10000) {
out += static_cast<char>(0xE0 | (cp >> 12));
out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
out += static_cast<char>(0x80 | (cp & 0x3F));
} else {
out += static_cast<char>(0xF0 | (cp >> 18));
out += static_cast<char>(0x80 | ((cp >> 12) & 0x3F));
out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
out += static_cast<char>(0x80 | (cp & 0x3F));
}
}
uint32_t parse_codepoint() {
if (pos_ + 4 > input_.size()) throw ParseError("short \\u escape");
uint32_t code = hex4(input_.substr(pos_, 4));
pos_ += 4;
// UTF-16 surrogate pair handling.
if (code >= 0xD800 && code <= 0xDBFF
&& pos_ + 6 <= input_.size()
&& input_.compare(pos_, 2, "\\u") == 0) {
const uint32_t lo = hex4(input_.substr(pos_ + 2, 4));
if (lo >= 0xDC00 && lo <= 0xDFFF) {
pos_ += 6;
return 0x10000 + ((code - 0xD800) << 10) + (lo - 0xDC00);
}
}
if (code > 0x10FFFF) throw ParseError("invalid unicode codepoint");
return code;
}
uint32_t hex4(const std::string& s) const {
uint32_t v = 0;
for (char c : s) {
v <<= 4;
if (c >= '0' && c <= '9') v |= static_cast<uint32_t>(c - '0');
else if (c >= 'a' && c <= 'f') v |= static_cast<uint32_t>(c - 'a' + 10);
else if (c >= 'A' && c <= 'F') v |= static_cast<uint32_t>(c - 'A' + 10);
else throw ParseError("bad \\u escape");
}
return v;
}
Value parse_bool() {
if (input_.compare(pos_, 4, "true") == 0) {
pos_ += 4;
return Value::boolean(true);
}
if (input_.compare(pos_, 5, "false") == 0) {
pos_ += 5;
return Value::boolean(false);
}
throw ParseError("invalid literal");
}
Value parse_null() {
if (input_.compare(pos_, 4, "null") == 0) {
pos_ += 4;
return Value::null();
}
throw ParseError("invalid literal");
}
Value parse_number() {
const size_t start = pos_;
if (peek() == '-') pos_++;
while (pos_ < input_.size()) {
const char c = input_[pos_];
if ((c >= '0' && c <= '9') || c == '.' || c == 'e' || c == 'E' || c == '+' || c == '-') {
pos_++;
} else {
break;
}
}
try {
return Value::number(std::stod(input_.substr(start, pos_ - start)));
} catch (const std::exception&) {
throw ParseError("bad number " + input_.substr(start, pos_ - start));
}
}
public:
// Position of the next unread byte — used by parse_json's trailing check.
size_t position() const { return pos_; }
void skip_trailing_ws() { skip_ws(); }
};
/// Parse a JSON document into a Value. Throws ParseError on bad input.
/// Callers holding a raw JSON string enter here.
inline Value parse_json(const std::string& input) {
Parser p(input);
Value v = p.parse_value();
p.skip_trailing_ws();
if (p.position() != input.size()) {
throw ParseError("trailing data at byte " + std::to_string(p.position()));
}
return v;
}
// ---------------------------------------------------------------------------
// JS-equivalent value semantics
// ---------------------------------------------------------------------------
// The canonical lib uses `typeof x === 'object'` and Object.keys(x), which in
// JavaScript treat BOTH objects and arrays as "object" and expose array indices
// as string keys ("0", "1", ...). We mirror that so degenerate inputs (e.g. an
// array of arrays) produce byte-identical output to the TS.
/// Object.keys parity: array indices as strings ("0", "1", ...), or the
/// object's insertion-ordered keys. Primitives and null yield no keys.
inline std::vector<std::string> keys_of(const Value& v) {
std::vector<std::string> keys;
switch (v.kind) {
case Value::Kind::Array:
for (size_t i = 0; i < v.items.size(); i++) keys.push_back(std::to_string(i));
break;
case Value::Kind::Object:
for (const auto& p : v.pairs) keys.push_back(p.first);
break;
default: break;
}
return keys;
}
/// JS `obj[key]` parity: object lookup, or array element at a non-negative
/// integer index. Returns nullptr when absent (which renders as the empty field).
inline const Value* get_field(const Value& v, const std::string& key) {
switch (v.kind) {
case Value::Kind::Object:
for (const auto& p : v.pairs) {
if (p.first == key) return &p.second;
}
return nullptr;
case Value::Kind::Array: {
if (key.empty() || key.size() > 19
|| key.find_first_not_of("0123456789") != std::string::npos) {
return nullptr;
}
const unsigned long long i = std::strtoull(key.c_str(), nullptr, 10);
return i < v.items.size() ? &v.items[i] : nullptr;
}
default:
return nullptr;
}
}
/// Render a number the way JS String(number) does on common inputs: shortest
/// decimal form that round-trips, with integral floats printed without a
/// trailing ".0" (e.g. `30.0` -> "30"). %.17g alone overshoots (0.1 would
/// print 17 digits), so walk precisions up until the value round-trips.
inline std::string format_number(double f) {
char buf[40];
for (int prec = 1; prec <= 17; prec++) {
const int n = std::snprintf(buf, sizeof(buf), "%.*g", prec, f);
const std::string s(buf, n > 0 ? static_cast<size_t>(n) : 0);
if (std::strtod(s.c_str(), nullptr) == f) return s;
}
return std::string(buf);
}
/// Coerce a JSON value to its display string, replicating JavaScript's
/// String(): null -> "", booleans -> "true"/"false", numbers -> decimal form,
/// arrays -> elements joined by "," (so a comma-bearing cell re-quotes), and
/// objects -> "[object Object]".
inline std::string js_string(const Value& v) {
switch (v.kind) {
case Value::Kind::Null: return "";
case Value::Kind::Bool: return v.b ? "true" : "false";
case Value::Kind::Number: return format_number(v.num);
case Value::Kind::Str: return v.str;
case Value::Kind::Array: {
std::string out;
for (size_t i = 0; i < v.items.size(); i++) {
if (i > 0) out += ',';
out += js_string(v.items[i]);
}
return out;
}
case Value::Kind::Object: return "[object Object]";
}
return "";
}
/// Quote a single CSV field per RFC 4180.
inline std::string csv_escape(const Value& v) {
const std::string s = js_string(v);
const bool needs_quoting = s.find_first_of(",\"\n\r") != std::string::npos;
if (!needs_quoting) return s;
std::string out = "\"";
for (char c : s) {
if (c == '"') out += "\"\"";
else out += c;
}
out += '"';
return out;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/// One deserialized CSV record: an ordered list of (header, cell) pairs. We use
/// a vector rather than a map so duplicate/empty headers survive round-trips,
/// exactly as in the TS lib's `Record<string, string>` indexing.
using CsvRow = std::vector<std::pair<std::string, std::string>>;
/// Serialize a JSON document to CSV.
///
/// Returns the CSV on success, std::nullopt when the input is not valid JSON
/// or when the document yields no object rows (and thus no headers) — e.g.
/// `[1, 2, 3]`. Accepts a single object or an array of objects.
inline std::optional<std::string> json_to_csv(const std::string& input) {
Value data;
try {
data = parse_json(input);
} catch (const ParseError&) {
return std::nullopt;
}
// A bare value is treated as a one-row table.
std::vector<const Value*> rows;
if (data.kind == Value::Kind::Array) {
for (const Value& item : data.items) rows.push_back(&item);
} else {
rows.push_back(&data);
}
// Header union across object-like rows, first-seen order, de-duplicated.
std::vector<std::string> headers;
std::unordered_set<std::string> seen;
for (const Value* row : rows) {
for (const std::string& k : keys_of(*row)) {
if (seen.insert(k).second) headers.push_back(k);
}
}
if (headers.empty()) return std::nullopt;
std::vector<std::string> lines;
lines.reserve(rows.size() + 1);
{
std::string header_line;
for (size_t i = 0; i < headers.size(); i++) {
if (i > 0) header_line += ',';
header_line += csv_escape(Value::string(headers[i]));
}
lines.push_back(std::move(header_line));
}
for (const Value* row : rows) {
// A non-object row (null, number, string) yields an empty line: every
// header lookup on it returns nullptr -> the empty field.
std::string line;
for (size_t i = 0; i < headers.size(); i++) {
if (i > 0) line += ',';
const Value* cell = get_field(*row, headers[i]);
line += csv_escape(cell ? *cell : Value::null());
}
lines.push_back(std::move(line));
}
std::string out;
for (size_t i = 0; i < lines.size(); i++) {
if (i > 0) out += '\n';
out += lines[i];
}
return out;
}
/// Parse RFC 4180 CSV into a vector of rows keyed by the first row.
///
/// Handles quoted fields, doubled-quote escapes, and embedded
/// commas/newlines; bare carriage returns outside quotes are ignored. Returns
/// an empty vector for empty input, or for input that is only a header row.
inline std::vector<CsvRow> csv_to_json(const std::string& csv) {
// Single-pass character-state machine over bytes. Field content outside
// ASCII passes through untouched (UTF-8 is transparent to the machine).
std::vector<std::vector<std::string>> rows;
std::string field;
std::vector<std::string> row;
bool in_quotes = false;
const size_t n = csv.size();
for (size_t i = 0; i < n; i++) {
const char ch = csv[i];
if (in_quotes) {
if (ch == '"') {
// Doubled quote -> one literal quote; lone quote -> close field.
if (i + 1 < n && csv[i + 1] == '"') {
field += '"';
i++;
continue;
}
in_quotes = false;
} else {
field += ch;
}
} else if (ch == '"') {
in_quotes = true;
} else if (ch == ',') {
row.push_back(std::move(field));
field.clear();
} else if (ch == '\n') {
row.push_back(std::move(field));
field.clear();
rows.push_back(std::move(row));
row.clear();
} else if (ch != '\r') {
field += ch;
}
}
// Flush a trailing row only when there is pending content. Input that ended
// with a newline already flushed; this guard avoids an empty final row.
if (!field.empty() || !row.empty()) {
row.push_back(std::move(field));
rows.push_back(std::move(row));
}
if (rows.empty()) return {};
const std::vector<std::string>& headers = rows[0];
std::vector<CsvRow> out;
out.reserve(rows.size() - 1);
for (size_t r = 1; r < rows.size(); r++) {
CsvRow rec;
rec.reserve(headers.size());
for (size_t i = 0; i < headers.size(); i++) {
rec.emplace_back(headers[i], i < rows[r].size() ? rows[r][i] : "");
}
out.push_back(std::move(rec));
}
return out;
}
} // namespace jsoncsv
int main() {
// Small end-to-end demo so this file is runnable as a showcase.
using jsoncsv::csv_to_json;
using jsoncsv::json_to_csv;
const std::string raw =
"[{\"name\":\"Doe, John\",\"note\":\"say \\\"hi\\\"\"},{\"name\":\"Jane\",\"note\":\"plain\"}]";
const auto csv = json_to_csv(raw);
if (!csv) {
std::fprintf(stderr, "error: no CSV produced\n");
return 1;
}
std::printf("%s\n", csv->c_str());
for (const auto& row : csv_to_json(*csv)) {
std::printf("{");
for (size_t i = 0; i < row.size(); i++) {
std::printf("%s%s=\"%s\"", i > 0 ? ", " : "", row[i].first.c_str(),
row[i].second.c_str());
}
std::printf("}\n");
}
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →