Skip to content

JSON ↔ CSV Converter — C++ source

Convert a JSON array of objects to CSV and back. Handles quoted fields, embedded commas, newlines and escaped quotes (RFC 4180). 100% in-browser.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// =============================================================================
// json-csv — C++ port
// =============================================================================
// Convert between JSON and RFC 4180 CSV in either direction:
//   • json_to_csv — serialize a JSON document (object or array of objects) to CSV
//   • csv_to_json — parse RFC 4180 CSV (with quoting) into a vector of row objects
//
// Language: C++17 — standard library only; minimal embedded JSON parser.
// Source: CosmoDev polyglot showcase port of json-csv,
//         ported from src/lib/csv.ts (the canonical, live TypeScript lib).
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Pure and deterministic — depends only on its inputs. RFC 4180 quoting: any
// field containing a comma, double quote, carriage return, or line feed is
// wrapped in double quotes, and each embedded quote is doubled.
//
// This is display source — part of CosmoDev's polyglot tool pages.
// =============================================================================

// C++'s stdlib has no JSON support, so this file is self-contained: a small
// recursive-descent parser, mirroring the Rust sibling snippet (which would
// use an ecosystem crate such as `serde_json` instead). Object pairs live in
// a vector so key order — which is observable (it determines CSV column
// order) — is preserved, exactly as in the canonical lib.

#include <cctype>
#include <cstdint>
#include <cstdio>
#include <cstdlib>
#include <optional>
#include <stdexcept>
#include <string>
#include <unordered_set>
#include <utility>
#include <vector>

namespace jsoncsv {

// ---------------------------------------------------------------------------
// JSON value tree
// ---------------------------------------------------------------------------

struct Value {
    enum class Kind { Null, Bool, Number, Str, Array, Object };

    Kind kind = Kind::Null;
    bool b = false;
    double num = 0.0;
    std::string str;
    std::vector<Value> items;                 // Array
    std::vector<std::pair<std::string, Value>> pairs;  // Object, insertion-ordered

    static Value null() { return Value{}; }
    static Value boolean(bool v) { Value x; x.kind = Kind::Bool; x.b = v; return x; }
    static Value number(double v) { Value x; x.kind = Kind::Number; x.num = v; return x; }
    static Value string(std::string v) { Value x; x.kind = Kind::Str; x.str = std::move(v); return x; }
    static Value array(std::vector<Value> v) { Value x; x.kind = Kind::Array; x.items = std::move(v); return x; }
    static Value object(std::vector<std::pair<std::string, Value>> v) { Value x; x.kind = Kind::Object; x.pairs = std::move(v); return x; }
};

// ---------------------------------------------------------------------------
// Minimal JSON parser
// ---------------------------------------------------------------------------
// Compact recursive-descent parser. Sufficient for any RFC 8259 document a
// caller is likely to feed this tool.

class ParseError : public std::runtime_error {
public:
    explicit ParseError(const std::string& msg) : std::runtime_error(msg) {}
};

class Parser {
public:
    explicit Parser(const std::string& input) : input_(input) {}

    Value parse_value() {
        skip_ws();
        require_more("unexpected end of input");
        const char c = peek();
        switch (c) {
            case '{': return parse_object();
            case '[': return parse_array();
            case '"': return Value::string(parse_string());
            case 't': case 'f': return parse_bool();
            case 'n': return parse_null();
            default:
                if (c == '-' || (c >= '0' && c <= '9')) return parse_number();
                throw ParseError(std::string("unexpected character '") + c + "'");
        }
    }

private:
    const std::string& input_;
    size_t pos_ = 0;

    void skip_ws() {
        while (pos_ < input_.size()) {
            switch (input_[pos_]) {
                case ' ': case '\t': case '\n': case '\r': pos_++; break;
                default: return;
            }
        }
    }

    void require_more(const char* msg) const {
        if (pos_ >= input_.size()) throw ParseError(msg);
    }

    char peek() const { return input_[pos_]; }

    Value parse_object() {
        pos_++; // {
        std::vector<std::pair<std::string, Value>> pairs;
        skip_ws();
        require_more("expected key or '}' in object");
        if (peek() == '}') {
            pos_++;
            return Value::object(std::move(pairs));
        }
        while (true) {
            skip_ws();
            if (peek() != '"') throw ParseError("expected string key in object");
            std::string key = parse_string();
            skip_ws();
            if (peek() != ':') throw ParseError("expected ':' after object key");
            pos_++;
            Value val = parse_value();
            // First occurrence of a key wins, matching JS object-literal semantics.
            bool dup = false;
            for (const auto& p : pairs) {
                if (p.first == key) { dup = true; break; }
            }
            if (!dup) pairs.emplace_back(std::move(key), std::move(val));
            skip_ws();
            switch (peek()) {
                case ',': pos_++; break;
                case '}': pos_++; return Value::object(std::move(pairs));
                default: throw ParseError("expected ',' or '}' in object");
            }
        }
    }

    Value parse_array() {
        pos_++; // [
        std::vector<Value> items;
        skip_ws();
        require_more("expected value or ']' in array");
        if (peek() == ']') {
            pos_++;
            return Value::array(std::move(items));
        }
        while (true) {
            items.push_back(parse_value());
            skip_ws();
            switch (peek()) {
                case ',': pos_++; break;
                case ']': pos_++; return Value::array(std::move(items));
                default: throw ParseError("expected ',' or ']' in array");
            }
        }
    }

    std::string parse_string() {
        pos_++; // opening quote
        std::string out;
        while (pos_ < input_.size()) {
            const char c = input_[pos_++];
            if (c == '"') return out;
            if (c == '\\') {
                require_more("trailing escape");
                const char e = input_[pos_++];
                switch (e) {
                    case '"': out += '"'; break;
                    case '\\': out += '\\'; break;
                    case '/': out += '/'; break;
                    case 'n': out += '\n'; break;
                    case 't': out += '\t'; break;
                    case 'r': out += '\r'; break;
                    case 'b': out += '\b'; break;
                    case 'f': out += '\f'; break;
                    case 'u': append_codepoint(out, parse_codepoint()); break;
                    default: throw ParseError(std::string("bad escape \\") + e);
                }
            } else {
                out += c;
            }
        }
        throw ParseError("unterminated string");
    }

    // Encode a codepoint as UTF-8 (JSON strings decode to UTF-8 in every
    // sibling port; \u escapes outside the BMP arrive as surrogate pairs).
    static void append_codepoint(std::string& out, uint32_t cp) {
        if (cp < 0x80) {
            out += static_cast<char>(cp);
        } else if (cp < 0x800) {
            out += static_cast<char>(0xC0 | (cp >> 6));
            out += static_cast<char>(0x80 | (cp & 0x3F));
        } else if (cp < 0x10000) {
            out += static_cast<char>(0xE0 | (cp >> 12));
            out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
            out += static_cast<char>(0x80 | (cp & 0x3F));
        } else {
            out += static_cast<char>(0xF0 | (cp >> 18));
            out += static_cast<char>(0x80 | ((cp >> 12) & 0x3F));
            out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
            out += static_cast<char>(0x80 | (cp & 0x3F));
        }
    }

    uint32_t parse_codepoint() {
        if (pos_ + 4 > input_.size()) throw ParseError("short \\u escape");
        uint32_t code = hex4(input_.substr(pos_, 4));
        pos_ += 4;
        // UTF-16 surrogate pair handling.
        if (code >= 0xD800 && code <= 0xDBFF
                && pos_ + 6 <= input_.size()
                && input_.compare(pos_, 2, "\\u") == 0) {
            const uint32_t lo = hex4(input_.substr(pos_ + 2, 4));
            if (lo >= 0xDC00 && lo <= 0xDFFF) {
                pos_ += 6;
                return 0x10000 + ((code - 0xD800) << 10) + (lo - 0xDC00);
            }
        }
        if (code > 0x10FFFF) throw ParseError("invalid unicode codepoint");
        return code;
    }

    uint32_t hex4(const std::string& s) const {
        uint32_t v = 0;
        for (char c : s) {
            v <<= 4;
            if (c >= '0' && c <= '9') v |= static_cast<uint32_t>(c - '0');
            else if (c >= 'a' && c <= 'f') v |= static_cast<uint32_t>(c - 'a' + 10);
            else if (c >= 'A' && c <= 'F') v |= static_cast<uint32_t>(c - 'A' + 10);
            else throw ParseError("bad \\u escape");
        }
        return v;
    }

    Value parse_bool() {
        if (input_.compare(pos_, 4, "true") == 0) {
            pos_ += 4;
            return Value::boolean(true);
        }
        if (input_.compare(pos_, 5, "false") == 0) {
            pos_ += 5;
            return Value::boolean(false);
        }
        throw ParseError("invalid literal");
    }

    Value parse_null() {
        if (input_.compare(pos_, 4, "null") == 0) {
            pos_ += 4;
            return Value::null();
        }
        throw ParseError("invalid literal");
    }

    Value parse_number() {
        const size_t start = pos_;
        if (peek() == '-') pos_++;
        while (pos_ < input_.size()) {
            const char c = input_[pos_];
            if ((c >= '0' && c <= '9') || c == '.' || c == 'e' || c == 'E' || c == '+' || c == '-') {
                pos_++;
            } else {
                break;
            }
        }
        try {
            return Value::number(std::stod(input_.substr(start, pos_ - start)));
        } catch (const std::exception&) {
            throw ParseError("bad number " + input_.substr(start, pos_ - start));
        }
    }
public:
    // Position of the next unread byte — used by parse_json's trailing check.
    size_t position() const { return pos_; }
    void skip_trailing_ws() { skip_ws(); }
};

/// Parse a JSON document into a Value. Throws ParseError on bad input.
/// Callers holding a raw JSON string enter here.
inline Value parse_json(const std::string& input) {
    Parser p(input);
    Value v = p.parse_value();
    p.skip_trailing_ws();
    if (p.position() != input.size()) {
        throw ParseError("trailing data at byte " + std::to_string(p.position()));
    }
    return v;
}

// ---------------------------------------------------------------------------
// JS-equivalent value semantics
// ---------------------------------------------------------------------------
// The canonical lib uses `typeof x === 'object'` and Object.keys(x), which in
// JavaScript treat BOTH objects and arrays as "object" and expose array indices
// as string keys ("0", "1", ...). We mirror that so degenerate inputs (e.g. an
// array of arrays) produce byte-identical output to the TS.

/// Object.keys parity: array indices as strings ("0", "1", ...), or the
/// object's insertion-ordered keys. Primitives and null yield no keys.
inline std::vector<std::string> keys_of(const Value& v) {
    std::vector<std::string> keys;
    switch (v.kind) {
        case Value::Kind::Array:
            for (size_t i = 0; i < v.items.size(); i++) keys.push_back(std::to_string(i));
            break;
        case Value::Kind::Object:
            for (const auto& p : v.pairs) keys.push_back(p.first);
            break;
        default: break;
    }
    return keys;
}

/// JS `obj[key]` parity: object lookup, or array element at a non-negative
/// integer index. Returns nullptr when absent (which renders as the empty field).
inline const Value* get_field(const Value& v, const std::string& key) {
    switch (v.kind) {
        case Value::Kind::Object:
            for (const auto& p : v.pairs) {
                if (p.first == key) return &p.second;
            }
            return nullptr;
        case Value::Kind::Array: {
            if (key.empty() || key.size() > 19
                    || key.find_first_not_of("0123456789") != std::string::npos) {
                return nullptr;
            }
            const unsigned long long i = std::strtoull(key.c_str(), nullptr, 10);
            return i < v.items.size() ? &v.items[i] : nullptr;
        }
        default:
            return nullptr;
    }
}

/// Render a number the way JS String(number) does on common inputs: shortest
/// decimal form that round-trips, with integral floats printed without a
/// trailing ".0" (e.g. `30.0` -> "30"). %.17g alone overshoots (0.1 would
/// print 17 digits), so walk precisions up until the value round-trips.
inline std::string format_number(double f) {
    char buf[40];
    for (int prec = 1; prec <= 17; prec++) {
        const int n = std::snprintf(buf, sizeof(buf), "%.*g", prec, f);
        const std::string s(buf, n > 0 ? static_cast<size_t>(n) : 0);
        if (std::strtod(s.c_str(), nullptr) == f) return s;
    }
    return std::string(buf);
}

/// Coerce a JSON value to its display string, replicating JavaScript's
/// String(): null -> "", booleans -> "true"/"false", numbers -> decimal form,
/// arrays -> elements joined by "," (so a comma-bearing cell re-quotes), and
/// objects -> "[object Object]".
inline std::string js_string(const Value& v) {
    switch (v.kind) {
        case Value::Kind::Null: return "";
        case Value::Kind::Bool: return v.b ? "true" : "false";
        case Value::Kind::Number: return format_number(v.num);
        case Value::Kind::Str: return v.str;
        case Value::Kind::Array: {
            std::string out;
            for (size_t i = 0; i < v.items.size(); i++) {
                if (i > 0) out += ',';
                out += js_string(v.items[i]);
            }
            return out;
        }
        case Value::Kind::Object: return "[object Object]";
    }
    return "";
}

/// Quote a single CSV field per RFC 4180.
inline std::string csv_escape(const Value& v) {
    const std::string s = js_string(v);
    const bool needs_quoting = s.find_first_of(",\"\n\r") != std::string::npos;
    if (!needs_quoting) return s;
    std::string out = "\"";
    for (char c : s) {
        if (c == '"') out += "\"\"";
        else out += c;
    }
    out += '"';
    return out;
}

// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------

/// One deserialized CSV record: an ordered list of (header, cell) pairs. We use
/// a vector rather than a map so duplicate/empty headers survive round-trips,
/// exactly as in the TS lib's `Record<string, string>` indexing.
using CsvRow = std::vector<std::pair<std::string, std::string>>;

/// Serialize a JSON document to CSV.
///
/// Returns the CSV on success, std::nullopt when the input is not valid JSON
/// or when the document yields no object rows (and thus no headers) — e.g.
/// `[1, 2, 3]`. Accepts a single object or an array of objects.
inline std::optional<std::string> json_to_csv(const std::string& input) {
    Value data;
    try {
        data = parse_json(input);
    } catch (const ParseError&) {
        return std::nullopt;
    }

    // A bare value is treated as a one-row table.
    std::vector<const Value*> rows;
    if (data.kind == Value::Kind::Array) {
        for (const Value& item : data.items) rows.push_back(&item);
    } else {
        rows.push_back(&data);
    }

    // Header union across object-like rows, first-seen order, de-duplicated.
    std::vector<std::string> headers;
    std::unordered_set<std::string> seen;
    for (const Value* row : rows) {
        for (const std::string& k : keys_of(*row)) {
            if (seen.insert(k).second) headers.push_back(k);
        }
    }
    if (headers.empty()) return std::nullopt;

    std::vector<std::string> lines;
    lines.reserve(rows.size() + 1);
    {
        std::string header_line;
        for (size_t i = 0; i < headers.size(); i++) {
            if (i > 0) header_line += ',';
            header_line += csv_escape(Value::string(headers[i]));
        }
        lines.push_back(std::move(header_line));
    }
    for (const Value* row : rows) {
        // A non-object row (null, number, string) yields an empty line: every
        // header lookup on it returns nullptr -> the empty field.
        std::string line;
        for (size_t i = 0; i < headers.size(); i++) {
            if (i > 0) line += ',';
            const Value* cell = get_field(*row, headers[i]);
            line += csv_escape(cell ? *cell : Value::null());
        }
        lines.push_back(std::move(line));
    }

    std::string out;
    for (size_t i = 0; i < lines.size(); i++) {
        if (i > 0) out += '\n';
        out += lines[i];
    }
    return out;
}

/// Parse RFC 4180 CSV into a vector of rows keyed by the first row.
///
/// Handles quoted fields, doubled-quote escapes, and embedded
/// commas/newlines; bare carriage returns outside quotes are ignored. Returns
/// an empty vector for empty input, or for input that is only a header row.
inline std::vector<CsvRow> csv_to_json(const std::string& csv) {
    // Single-pass character-state machine over bytes. Field content outside
    // ASCII passes through untouched (UTF-8 is transparent to the machine).
    std::vector<std::vector<std::string>> rows;
    std::string field;
    std::vector<std::string> row;
    bool in_quotes = false;

    const size_t n = csv.size();
    for (size_t i = 0; i < n; i++) {
        const char ch = csv[i];
        if (in_quotes) {
            if (ch == '"') {
                // Doubled quote -> one literal quote; lone quote -> close field.
                if (i + 1 < n && csv[i + 1] == '"') {
                    field += '"';
                    i++;
                    continue;
                }
                in_quotes = false;
            } else {
                field += ch;
            }
        } else if (ch == '"') {
            in_quotes = true;
        } else if (ch == ',') {
            row.push_back(std::move(field));
            field.clear();
        } else if (ch == '\n') {
            row.push_back(std::move(field));
            field.clear();
            rows.push_back(std::move(row));
            row.clear();
        } else if (ch != '\r') {
            field += ch;
        }
    }

    // Flush a trailing row only when there is pending content. Input that ended
    // with a newline already flushed; this guard avoids an empty final row.
    if (!field.empty() || !row.empty()) {
        row.push_back(std::move(field));
        rows.push_back(std::move(row));
    }

    if (rows.empty()) return {};

    const std::vector<std::string>& headers = rows[0];
    std::vector<CsvRow> out;
    out.reserve(rows.size() - 1);
    for (size_t r = 1; r < rows.size(); r++) {
        CsvRow rec;
        rec.reserve(headers.size());
        for (size_t i = 0; i < headers.size(); i++) {
            rec.emplace_back(headers[i], i < rows[r].size() ? rows[r][i] : "");
        }
        out.push_back(std::move(rec));
    }
    return out;
}

}  // namespace jsoncsv

int main() {
    // Small end-to-end demo so this file is runnable as a showcase.
    using jsoncsv::csv_to_json;
    using jsoncsv::json_to_csv;

    const std::string raw =
        "[{\"name\":\"Doe, John\",\"note\":\"say \\\"hi\\\"\"},{\"name\":\"Jane\",\"note\":\"plain\"}]";
    const auto csv = json_to_csv(raw);
    if (!csv) {
        std::fprintf(stderr, "error: no CSV produced\n");
        return 1;
    }
    std::printf("%s\n", csv->c_str());
    for (const auto& row : csv_to_json(*csv)) {
        std::printf("{");
        for (size_t i = 0; i < row.size(); i++) {
            std::printf("%s%s=\"%s\"", i > 0 ? ", " : "", row[i].first.c_str(),
                        row[i].second.c_str());
        }
        std::printf("}\n");
    }
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →