JSON ↔ CSV Converter — Python source
Convert a JSON array of objects to CSV and back. Handles quoted fields, embedded commas, newlines and escaped quotes (RFC 4180). 100% in-browser.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
#!/usr/bin/env python3
# =============================================================================
# json-csv — Python port
# =============================================================================
# Convert between JSON and RFC 4180 CSV in either direction:
# • json_to_csv — serialize a JSON document (object or list of objects) to CSV
# • csv_to_json — parse RFC 4180 CSV (with quoting) into a list of dicts
#
# CosmoDev polyglot showcase port of the `json-csv` tool.
# Ported from src/lib/csv.ts (the canonical, live TypeScript lib).
#
# Pure and deterministic — depends only on its inputs. RFC 4180 quoting: any
# field containing a comma, double quote, carriage return, or line feed is
# wrapped in double quotes, and each embedded quote is doubled ("").
#
# This is display source — part of CosmoDev's polyglot tool pages.
# =============================================================================
"""JSON <-> CSV conversion (RFC 4180 quoting), mirroring ``src/lib/csv.ts``."""
from __future__ import annotations
import json
import re
from typing import Any, List, Optional
__all__ = ["json_to_csv", "csv_to_json"]
# A field must be quoted when it contains any of: comma, double quote, CR, LF.
_NEEDS_QUOTING = re.compile(r'[",\n\r]')
# ---------------------------------------------------------------------------
# Number formatting (JS String(number) parity)
# ---------------------------------------------------------------------------
# Python's ``str(30.0)`` is ``"30.0"`` whereas JavaScript's ``String(30.0)`` is
# ``"30"``. CSV data is overwhelmingly string-typed, but to stay functionally
# equivalent on numeric values we drop a trailing ``.0`` on integral floats.
# ``repr`` gives the shortest round-trip form for non-integral floats.
def _num_str(n: float) -> str:
if isinstance(n, bool): # bool is an int subclass; handled by callers, but guard anyway
return "true" if n else "false"
if isinstance(n, int):
return str(n)
if n == int(n) and abs(n) < 1e16:
return str(int(n))
return repr(n)
def _js_string(value: Any) -> str:
"""Coerce a JSON value to its display string, replicating JS ``String()``.
``None`` -> "", booleans -> "true"/"false", numbers -> decimal form,
lists -> elements joined by "," (so a comma-bearing cell re-quotes), and
dicts -> "[object Object]".
"""
if value is None:
return ""
if isinstance(value, bool):
return "true" if value else "false"
if isinstance(value, (int, float)):
return _num_str(value)
if isinstance(value, str):
return value
if isinstance(value, list):
return ",".join(_js_string(e) for e in value)
# dict (or any other object) -> JS's [object Object].
return "[object Object]"
def _csv_escape(field: Any) -> str:
"""Quote a single CSV field per RFC 4180."""
s = _js_string(field)
if _NEEDS_QUOTING.search(s):
return '"' + s.replace('"', '""') + '"'
return s
# ---------------------------------------------------------------------------
# JS-equivalent value semantics
# ---------------------------------------------------------------------------
# The canonical lib uses ``typeof x === 'object'`` and ``Object.keys(x)``, which
# in JavaScript treat BOTH objects and arrays as "object" and expose array
# indices as string keys ("0", "1", ...). We mirror that so degenerate inputs
# (e.g. a list of lists) produce byte-identical output to the TS.
def _is_object_like(value: Any) -> bool:
"""True for dict and list (JS ``typeof === 'object' && != null``)."""
return isinstance(value, (dict, list))
def _keys(value: Any) -> List[str]:
"""Object.keys parity: list indices as strings, or dict keys in order."""
if isinstance(value, dict):
return list(value.keys())
if isinstance(value, list):
return [str(i) for i in range(len(value))]
return []
def _get(value: Any, key: str) -> Any:
"""JS ``obj[key]`` parity: dict lookup, or list element at a non-negative
integer index. Returns None when absent (which renders as the empty field).
"""
if isinstance(value, dict):
return value.get(key)
if isinstance(value, list):
try:
idx = int(key)
except (ValueError, TypeError):
return None
if 0 <= idx < len(value):
return value[idx]
return None
return None
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def json_to_csv(text: str) -> Optional[str]:
"""Serialize a JSON document to CSV.
Accepts a single object or a list of objects. Returns ``None`` on invalid
JSON, or when the document yields no object rows (and thus no column
headers) — e.g. a bare list of primitives such as ``[1, 2, 3]``.
"""
try:
data = json.loads(text)
except (ValueError, TypeError):
return None
# A bare value is treated as a one-row table.
rows = data if isinstance(data, list) else [data]
# Header union across object-like rows, first-seen order, de-duplicated.
headers: List[str] = []
seen = set()
for row in rows:
for k in _keys(row):
if k not in seen:
seen.add(k)
headers.append(k)
if not headers:
return None
# First line is the (escaped) header row; subsequent lines are the rows.
lines = [",".join(_csv_escape(h) for h in headers)]
for row in rows:
# A non-object row (None, number, string) yields an empty line: every
# header lookup on it returns None -> the empty field.
cells = [_csv_escape(_get(row, h)) for h in headers]
lines.append(",".join(cells))
return "\n".join(lines)
def csv_to_json(text: str) -> List[dict]:
"""Parse RFC 4180 CSV into a list of row dicts keyed by the first row.
Handles quoted fields, doubled-quote escapes, and embedded
commas/newlines; bare carriage returns outside quotes are ignored. Returns
``[]`` for empty input, or for input that is only a header row.
"""
# Single-pass character-state machine. ``text`` is indexed by position so
# we can look one character ahead for the doubled-quote escape.
rows: List[List[str]] = []
field = []
row: List[str] = []
in_quotes = False
n = len(text)
i = 0
while i < n:
ch = text[i]
if in_quotes:
if ch == '"':
# Doubled quote -> one literal quote; lone quote -> close field.
if i + 1 < n and text[i + 1] == '"':
field.append('"')
i += 2
continue
in_quotes = False
else:
field.append(ch)
elif ch == '"':
in_quotes = True
elif ch == ',':
row.append("".join(field))
field = []
elif ch == '\n':
row.append("".join(field))
rows.append(row)
row = []
field = []
elif ch != '\r':
field.append(ch)
i += 1
# Flush a trailing row only when there is pending content. Input that ended
# with a newline already flushed; this guard avoids an empty final row.
if field or row:
row.append("".join(field))
rows.append(row)
if not rows:
return []
headers = rows[0]
out: List[dict] = []
for r in rows[1:]:
obj = {h: (r[i] if i < len(r) else "") for i, h in enumerate(headers)}
out.append(obj)
return out
if __name__ == "__main__":
# Small end-to-end demo so this file is runnable as a showcase.
raw = '[{"name":"Doe, John","note":"say \\"hi\\""},{"name":"Jane","note":"plain"}]'
csv = json_to_csv(raw)
print(csv)
print(csv_to_json(csv))
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →