JSON ↔ CSV Converter — Rust source
Convert a JSON array of objects to CSV and back. Handles quoted fields, embedded commas, newlines and escaped quotes (RFC 4180). 100% in-browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// =============================================================================
// json-csv — Rust port
// =============================================================================
// Convert between JSON and RFC 4180 CSV in either direction:
// • json_to_csv — serialize a JSON document (object or array of objects) to CSV
// • csv_to_json — parse RFC 4180 CSV (with quoting) into a Vec of row objects
//
// CosmoDev polyglot showcase port of the `json-csv` tool.
// Ported from src/lib/csv.ts (the canonical, live TypeScript lib).
//
// Pure and deterministic — depends only on its inputs. RFC 4180 quoting: any
// field containing a comma, double quote, carriage return, or line feed is
// wrapped in double quotes, and each embedded quote is doubled.
//
// This is display source — part of CosmoDev's polyglot tool pages.
// =============================================================================
// Rust's stdlib has no JSON support, so this file is self-contained: a tiny
// `Value` enum plus a minimal recursive-descent parser. Object pairs live in a
// `Vec` so key order — which is observable (it determines CSV column order) — is
// preserved, exactly as in the canonical lib.
use std::collections::HashSet;
// ---------------------------------------------------------------------------
// JSON value tree
// ---------------------------------------------------------------------------
#[derive(Clone, Debug)]
pub enum Value {
Null,
Bool(bool),
Number(f64),
Str(String),
Array(Vec<Value>),
// Insertion-ordered (key, value) pairs; duplicates keep their first position.
Object(Vec<(String, Value)>),
}
// ---------------------------------------------------------------------------
// Minimal JSON parser
// ---------------------------------------------------------------------------
// Compact recursive-descent parser. Sufficient for any RFC 8259 document a
// caller is likely to feed this tool.
type PResult<T> = Result<T, String>;
struct Parser<'a> {
bytes: &'a [u8],
pos: usize,
}
impl<'a> Parser<'a> {
fn new(input: &'a str) -> Self {
Parser { bytes: input.as_bytes(), pos: 0 }
}
fn skip_ws(&mut self) {
while self.pos < self.bytes.len() {
match self.bytes[self.pos] {
b' ' | b'\t' | b'\n' | b'\r' => self.pos += 1,
_ => break,
}
}
}
fn peek(&self) -> Option<u8> {
self.bytes.get(self.pos).copied()
}
fn parse_value(&mut self) -> PResult<Value> {
self.skip_ws();
match self.peek().ok_or_else(|| "unexpected end of input".to_string())? {
b'{' => self.parse_object(),
b'[' => self.parse_array(),
b'"' => Ok(Value::Str(self.parse_string()?)),
b't' | b'f' => self.parse_bool(),
b'n' => self.parse_null(),
b'-' | b'0'..=b'9' => self.parse_number(),
c => Err(format!("unexpected character {:?}", c as char)),
}
}
fn parse_object(&mut self) -> PResult<Value> {
self.pos += 1; // {
let mut pairs: Vec<(String, Value)> = Vec::new();
self.skip_ws();
if self.peek() == Some(b'}') {
self.pos += 1;
return Ok(Value::Object(pairs));
}
loop {
self.skip_ws();
if self.peek() != Some(b'"') {
return Err("expected string key in object".to_string());
}
let key = self.parse_string()?;
self.skip_ws();
if self.peek() != Some(b':') {
return Err("expected ':' after object key".to_string());
}
self.pos += 1;
let val = self.parse_value()?;
// First occurrence of a key wins, matching JS object-literal semantics.
if !pairs.iter().any(|(k, _)| k == &key) {
pairs.push((key, val));
}
self.skip_ws();
match self.peek() {
Some(b',') => self.pos += 1,
Some(b'}') => {
self.pos += 1;
break;
}
_ => return Err("expected ',' or '}' in object".to_string()),
}
}
Ok(Value::Object(pairs))
}
fn parse_array(&mut self) -> PResult<Value> {
self.pos += 1; // [
let mut items = Vec::new();
self.skip_ws();
if self.peek() == Some(b']') {
self.pos += 1;
return Ok(Value::Array(items));
}
loop {
let val = self.parse_value()?;
items.push(val);
self.skip_ws();
match self.peek() {
Some(b',') => self.pos += 1,
Some(b']') => {
self.pos += 1;
break;
}
_ => return Err("expected ',' or ']' in array".to_string()),
}
}
Ok(Value::Array(items))
}
fn parse_string(&mut self) -> PResult<String> {
self.pos += 1; // opening quote
let mut out = String::new();
while let Some(c) = self.peek() {
self.pos += 1;
match c {
b'"' => return Ok(out),
b'\\' => {
let e = self.peek().ok_or_else(|| "trailing escape".to_string())?;
self.pos += 1;
match e {
b'"' => out.push('"'),
b'\\' => out.push('\\'),
b'/' => out.push('/'),
b'n' => out.push('\n'),
b't' => out.push('\t'),
b'r' => out.push('\r'),
b'b' => out.push('\u{0008}'),
b'f' => out.push('\u{000C}'),
b'u' => {
let cp = self.parse_codepoint()?;
out.push(cp);
}
_ => return Err(format!("bad escape \\{}", e as char)),
}
}
_ => out.push(c as char),
}
}
Err("unterminated string".to_string())
}
fn parse_codepoint(&mut self) -> PResult<char> {
if self.pos + 4 > self.bytes.len() {
return Err("short \\u escape".to_string());
}
let hex = std::str::from_utf8(&self.bytes[self.pos..self.pos + 4])
.map_err(|_| "non-ascii in \\u escape".to_string())?;
self.pos += 4;
let mut code =
u32::from_str_radix(hex, 16).map_err(|_| "bad \\u escape".to_string())?;
// UTF-16 surrogate pair handling.
if (0xD800..=0xDBFF).contains(&code) {
if self.bytes[self.pos..].starts_with(b"\\u") {
self.pos += 2;
let lo_hex = std::str::from_utf8(&self.bytes[self.pos..self.pos + 4])
.map_err(|_| "non-ascii in low surrogate".to_string())?;
self.pos += 4;
let lo = u32::from_str_radix(lo_hex, 16)
.map_err(|_| "bad low surrogate".to_string())?;
if (0xDC00..=0xDFFF).contains(&lo) {
code = 0x10000 + ((code - 0xD800) << 10) + (lo - 0xDC00);
}
}
}
char::from_u32(code).ok_or_else(|| "invalid unicode codepoint".to_string())
}
fn parse_bool(&mut self) -> PResult<Value> {
if self.bytes[self.pos..].starts_with(b"true") {
self.pos += 4;
Ok(Value::Bool(true))
} else if self.bytes[self.pos..].starts_with(b"false") {
self.pos += 5;
Ok(Value::Bool(false))
} else {
Err("invalid literal".to_string())
}
}
fn parse_null(&mut self) -> PResult<Value> {
if self.bytes[self.pos..].starts_with(b"null") {
self.pos += 4;
Ok(Value::Null)
} else {
Err("invalid literal".to_string())
}
}
fn parse_number(&mut self) -> PResult<Value> {
let start = self.pos;
if self.peek() == Some(b'-') {
self.pos += 1;
}
while let Some(c) = self.peek() {
match c {
b'0'..=b'9' | b'.' | b'e' | b'E' | b'+' | b'-' => self.pos += 1,
_ => break,
}
}
let s = std::str::from_utf8(&self.bytes[start..self.pos])
.map_err(|_| "non-utf8 number".to_string())?;
let n: f64 = s.parse().map_err(|_| format!("bad number {}", s))?;
Ok(Value::Number(n))
}
}
/// Parse a JSON document into a [Value]. Callers holding a raw JSON string
/// enter here.
pub fn parse_json(input: &str) -> PResult<Value> {
let mut p = Parser::new(input);
let v = p.parse_value()?;
p.skip_ws();
if p.pos != p.bytes.len() {
return Err(format!("trailing data at byte {}", p.pos));
}
Ok(v)
}
// ---------------------------------------------------------------------------
// JS-equivalent value semantics
// ---------------------------------------------------------------------------
// The canonical lib uses `typeof x === 'object'` and Object.keys(x), which in
// JavaScript treat BOTH objects and arrays as "object" and expose array indices
// as string keys ("0", "1", ...). We mirror that so degenerate inputs (e.g. an
// array of arrays) produce byte-identical output to the TS.
/// Object.keys parity: array indices as strings ("0", "1", ...), or the
/// object's insertion-ordered keys. Primitives and null yield no keys.
fn keys_of(v: &Value) -> Vec<String> {
match v {
Value::Array(items) => (0..items.len()).map(|i| i.to_string()).collect(),
Value::Object(pairs) => pairs.iter().map(|(k, _)| k.clone()).collect(),
_ => Vec::new(),
}
}
/// JS `obj[key]` parity: object lookup, or array element at a non-negative
/// integer index. Returns None when absent (which renders as the empty field).
fn get_field<'a>(v: &'a Value, key: &str) -> Option<&'a Value> {
match v {
Value::Object(pairs) => pairs.iter().find(|(k, _)| k == key).map(|(_, v)| v),
Value::Array(items) => key.parse::<usize>().ok().and_then(|i| items.get(i)),
_ => None,
}
}
/// Render a number the way JS String(number) does on common inputs. Rust's
/// default float Display already uses the shortest round-trip form and prints
/// integral floats without a trailing ".0" (e.g. `30.0` -> "30").
fn format_number(f: f64) -> String {
format!("{}", f)
}
/// Coerce a JSON value to its display string, replicating JavaScript's
/// String(): null -> "", booleans -> "true"/"false", numbers -> decimal form,
/// arrays -> elements joined by "," (so a comma-bearing cell re-quotes), and
/// objects -> "[object Object]".
fn js_string(v: &Value) -> String {
match v {
Value::Null => String::new(),
Value::Bool(b) => if *b { "true" } else { "false" }.to_string(),
Value::Number(n) => format_number(*n),
Value::Str(s) => s.clone(),
Value::Array(items) => {
items.iter().map(js_string).collect::<Vec<_>>().join(",")
}
Value::Object(_) => "[object Object]".to_string(),
}
}
/// Quote a single CSV field per RFC 4180.
fn csv_escape(v: &Value) -> String {
let s = js_string(v);
if s.chars().any(|c| matches!(c, ',' | '"' | '\n' | '\r')) {
format!("\"{}\"", s.replace('"', "\"\""))
} else {
s
}
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/// One deserialized CSV record: an ordered list of (header, cell) pairs. We use
/// a Vec rather than a HashMap so duplicate/empty headers survive round-trips,
/// exactly as in the TS lib's `Record<string, string>` indexing.
pub type CsvRow = Vec<(String, String)>;
/// Serialize a JSON document to CSV.
///
/// Returns `Ok(String)` on success, `Ok(None)` when the document yields no
/// object rows (and thus no headers) — e.g. `[1, 2, 3]` — and `Err` when the
/// input is not valid JSON. Accepts a single object or an array of objects.
pub fn json_to_csv(input: &str) -> Result<Option<String>, String> {
let data = parse_json(input)?;
// A bare value is treated as a one-row table.
let rows: Vec<&Value> = match &data {
Value::Array(items) => items.iter().collect(),
single => vec![single],
};
// Header union across object-like rows, first-seen order, de-duplicated.
let mut headers: Vec<String> = Vec::new();
let mut seen: HashSet<String> = HashSet::new();
for row in &rows {
for k in keys_of(row) {
if seen.insert(k.clone()) {
headers.push(k);
}
}
}
if headers.is_empty() {
return Ok(None);
}
let mut lines: Vec<String> = Vec::with_capacity(rows.len() + 1);
lines.push(
headers
.iter()
.map(|h| csv_escape(&Value::Str(h.clone())))
.collect::<Vec<_>>()
.join(","),
);
for row in rows {
// A non-object row (null, number, string) yields an empty line: every
// header lookup on it returns None -> the empty field.
let cells: Vec<String> = headers
.iter()
.map(|h| match get_field(row, h) {
Some(v) => csv_escape(v),
None => csv_escape(&Value::Null),
})
.collect();
lines.push(cells.join(","));
}
Ok(Some(lines.join("\n")))
}
/// Parse RFC 4180 CSV into a Vec of rows keyed by the first row.
///
/// Handles quoted fields, doubled-quote escapes, and embedded
/// commas/newlines; bare carriage returns outside quotes are ignored. Returns
/// an empty Vec for empty input, or for input that is only a header row.
pub fn csv_to_json(csv: &str) -> Vec<CsvRow> {
// Single-pass character-state machine over chars, so multi-byte field
// content is preserved. CSV structural characters are ASCII.
let mut rows: Vec<Vec<String>> = Vec::new();
let mut field = String::new();
let mut row: Vec<String> = Vec::new();
let mut in_quotes = false;
let chars: Vec<char> = csv.chars().collect();
let mut i = 0;
while i < chars.len() {
let ch = chars[i];
if in_quotes {
if ch == '"' {
// Doubled quote -> one literal quote; lone quote -> close field.
if i + 1 < chars.len() && chars[i + 1] == '"' {
field.push('"');
i += 2;
continue;
}
in_quotes = false;
} else {
field.push(ch);
}
} else if ch == '"' {
in_quotes = true;
} else if ch == ',' {
row.push(std::mem::take(&mut field));
} else if ch == '\n' {
row.push(std::mem::take(&mut field));
// mem::take swaps `row` with an empty Vec and moves the old one in.
rows.push(std::mem::take(&mut row));
} else if ch != '\r' {
field.push(ch);
}
i += 1;
}
// Flush a trailing row only when there is pending content. Input that ended
// with a newline already flushed; this guard avoids an empty final row.
if !field.is_empty() || !row.is_empty() {
row.push(field);
rows.push(row);
}
if rows.is_empty() {
return Vec::new();
}
let headers = &rows[0];
let mut out: Vec<CsvRow> = Vec::with_capacity(rows.len() - 1);
for r in &rows[1..] {
let obj: CsvRow = headers
.iter()
.enumerate()
.map(|(i, h)| (h.clone(), r.get(i).cloned().unwrap_or_default()))
.collect();
out.push(obj);
}
out
}
fn main() {
// Small end-to-end demo so this file is runnable as a showcase.
let raw = r#"[{"name":"Doe, John","note":"say \"hi\""},{"name":"Jane","note":"plain"}]"#;
match json_to_csv(raw) {
Ok(Some(csv)) => {
println!("{}", csv);
for row in csv_to_json(&csv) {
println!("{:?}", row);
}
}
Ok(None) => println!("(no CSV produced)"),
Err(e) => eprintln!("error: {}", e),
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn round_trips_quoting() {
let csv = json_to_csv(r#"[{"a":"1","b":"2"},{"a":"3","b":"4"}]"#)
.unwrap()
.unwrap();
assert_eq!(csv, "a,b\n1,2\n3,4");
let none = json_to_csv("[1,2,3]").unwrap();
assert_eq!(none, None);
let single = json_to_csv(r#"{"a":"1"}"#).unwrap().unwrap();
assert_eq!(single, "a\n1");
}
#[test]
fn parses_quoted_fields() {
let rows = csv_to_json("name\n\"Doe, John\"");
assert_eq!(rows.len(), 1);
assert_eq!(rows[0][0], ("name".to_string(), "Doe, John".to_string()));
assert!(csv_to_json("").is_empty());
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →