Skip to content

JSON ↔ CSV Converter — Python source

Convert a JSON array of objects to CSV and back. Handles quoted fields, embedded commas, newlines and escaped quotes (RFC 4180). 100% in-browser.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

#!/usr/bin/env python3
# =============================================================================
# json-csv — Python port
# =============================================================================
# Convert between JSON and RFC 4180 CSV in either direction:
#   • json_to_csv — serialize a JSON document (object or list of objects) to CSV
#   • csv_to_json — parse RFC 4180 CSV (with quoting) into a list of dicts
#
# CosmoDev polyglot showcase port of the `json-csv` tool.
# Ported from src/lib/csv.ts (the canonical, live TypeScript lib).
#
# Pure and deterministic — depends only on its inputs. RFC 4180 quoting: any
# field containing a comma, double quote, carriage return, or line feed is
# wrapped in double quotes, and each embedded quote is doubled ("").
#
# This is display source — part of CosmoDev's polyglot tool pages.
# =============================================================================

"""JSON <-> CSV conversion (RFC 4180 quoting), mirroring ``src/lib/csv.ts``."""

from __future__ import annotations

import json
import re
from typing import Any, List, Optional

__all__ = ["json_to_csv", "csv_to_json"]

# A field must be quoted when it contains any of: comma, double quote, CR, LF.
_NEEDS_QUOTING = re.compile(r'[",\n\r]')


# ---------------------------------------------------------------------------
# Number formatting (JS String(number) parity)
# ---------------------------------------------------------------------------
# Python's ``str(30.0)`` is ``"30.0"`` whereas JavaScript's ``String(30.0)`` is
# ``"30"``. CSV data is overwhelmingly string-typed, but to stay functionally
# equivalent on numeric values we drop a trailing ``.0`` on integral floats.
# ``repr`` gives the shortest round-trip form for non-integral floats.
def _num_str(n: float) -> str:
    if isinstance(n, bool):  # bool is an int subclass; handled by callers, but guard anyway
        return "true" if n else "false"
    if isinstance(n, int):
        return str(n)
    if n == int(n) and abs(n) < 1e16:
        return str(int(n))
    return repr(n)


def _js_string(value: Any) -> str:
    """Coerce a JSON value to its display string, replicating JS ``String()``.

    ``None`` -> "", booleans -> "true"/"false", numbers -> decimal form,
    lists -> elements joined by "," (so a comma-bearing cell re-quotes), and
    dicts -> "[object Object]".
    """
    if value is None:
        return ""
    if isinstance(value, bool):
        return "true" if value else "false"
    if isinstance(value, (int, float)):
        return _num_str(value)
    if isinstance(value, str):
        return value
    if isinstance(value, list):
        return ",".join(_js_string(e) for e in value)
    # dict (or any other object) -> JS's [object Object].
    return "[object Object]"


def _csv_escape(field: Any) -> str:
    """Quote a single CSV field per RFC 4180."""
    s = _js_string(field)
    if _NEEDS_QUOTING.search(s):
        return '"' + s.replace('"', '""') + '"'
    return s


# ---------------------------------------------------------------------------
# JS-equivalent value semantics
# ---------------------------------------------------------------------------
# The canonical lib uses ``typeof x === 'object'`` and ``Object.keys(x)``, which
# in JavaScript treat BOTH objects and arrays as "object" and expose array
# indices as string keys ("0", "1", ...). We mirror that so degenerate inputs
# (e.g. a list of lists) produce byte-identical output to the TS.


def _is_object_like(value: Any) -> bool:
    """True for dict and list (JS ``typeof === 'object' && != null``)."""
    return isinstance(value, (dict, list))


def _keys(value: Any) -> List[str]:
    """Object.keys parity: list indices as strings, or dict keys in order."""
    if isinstance(value, dict):
        return list(value.keys())
    if isinstance(value, list):
        return [str(i) for i in range(len(value))]
    return []


def _get(value: Any, key: str) -> Any:
    """JS ``obj[key]`` parity: dict lookup, or list element at a non-negative
    integer index. Returns None when absent (which renders as the empty field).
    """
    if isinstance(value, dict):
        return value.get(key)
    if isinstance(value, list):
        try:
            idx = int(key)
        except (ValueError, TypeError):
            return None
        if 0 <= idx < len(value):
            return value[idx]
        return None
    return None


# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------


def json_to_csv(text: str) -> Optional[str]:
    """Serialize a JSON document to CSV.

    Accepts a single object or a list of objects. Returns ``None`` on invalid
    JSON, or when the document yields no object rows (and thus no column
    headers) — e.g. a bare list of primitives such as ``[1, 2, 3]``.
    """
    try:
        data = json.loads(text)
    except (ValueError, TypeError):
        return None

    # A bare value is treated as a one-row table.
    rows = data if isinstance(data, list) else [data]

    # Header union across object-like rows, first-seen order, de-duplicated.
    headers: List[str] = []
    seen = set()
    for row in rows:
        for k in _keys(row):
            if k not in seen:
                seen.add(k)
                headers.append(k)
    if not headers:
        return None

    # First line is the (escaped) header row; subsequent lines are the rows.
    lines = [",".join(_csv_escape(h) for h in headers)]
    for row in rows:
        # A non-object row (None, number, string) yields an empty line: every
        # header lookup on it returns None -> the empty field.
        cells = [_csv_escape(_get(row, h)) for h in headers]
        lines.append(",".join(cells))
    return "\n".join(lines)


def csv_to_json(text: str) -> List[dict]:
    """Parse RFC 4180 CSV into a list of row dicts keyed by the first row.

    Handles quoted fields, doubled-quote escapes, and embedded
    commas/newlines; bare carriage returns outside quotes are ignored. Returns
    ``[]`` for empty input, or for input that is only a header row.
    """
    # Single-pass character-state machine. ``text`` is indexed by position so
    # we can look one character ahead for the doubled-quote escape.
    rows: List[List[str]] = []
    field = []
    row: List[str] = []
    in_quotes = False
    n = len(text)

    i = 0
    while i < n:
        ch = text[i]
        if in_quotes:
            if ch == '"':
                # Doubled quote -> one literal quote; lone quote -> close field.
                if i + 1 < n and text[i + 1] == '"':
                    field.append('"')
                    i += 2
                    continue
                in_quotes = False
            else:
                field.append(ch)
        elif ch == '"':
            in_quotes = True
        elif ch == ',':
            row.append("".join(field))
            field = []
        elif ch == '\n':
            row.append("".join(field))
            rows.append(row)
            row = []
            field = []
        elif ch != '\r':
            field.append(ch)
        i += 1

    # Flush a trailing row only when there is pending content. Input that ended
    # with a newline already flushed; this guard avoids an empty final row.
    if field or row:
        row.append("".join(field))
        rows.append(row)

    if not rows:
        return []

    headers = rows[0]
    out: List[dict] = []
    for r in rows[1:]:
        obj = {h: (r[i] if i < len(r) else "") for i, h in enumerate(headers)}
        out.append(obj)
    return out


if __name__ == "__main__":
    # Small end-to-end demo so this file is runnable as a showcase.
    raw = '[{"name":"Doe, John","note":"say \\"hi\\""},{"name":"Jane","note":"plain"}]'
    csv = json_to_csv(raw)
    print(csv)
    print(csv_to_json(csv))

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →