JSON to Zod Schema — Rust source
Generate Zod validation schemas from JSON. Infers z.string, z.number, z.boolean, z.object, z.array, z.null, and z.union for mixed arrays.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! json-to-zod — Rust polyglot showcase port.
//!
//! Recursively infers a Zod schema string from a JSON value. Mixed-type arrays
//! collapse to z.union(...); plain objects become z.object({...}); empty arrays
//! and objects fall back to z.array(z.unknown()) / z.object({}). The converter
//! never panics — JSON parse failures and inference problems are returned as
//! `Outcome { ok: false, error: Some(...) }`.
//!
//! Ported from src/lib/jsonToZod.ts (CosmoDev).
//! Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//!
//! Rust's standard library ships no JSON parser, so this file includes a small,
//! self-contained recursive-descent parser. Keeping it in-tree means the port
//! has zero external crates and full control over object key insertion order
//! (which a BTreeMap-based parser would sort away).
use std::collections::HashMap;
/// A JSON value. Object entries are stored in a `Vec<(String, Json)>` so the
/// parser preserves source insertion order — matching JavaScript's parser and
/// keeping generated field order stable.
#[derive(Debug, Clone)]
enum Json {
Null,
Bool(bool),
Num(f64),
Str(String),
Arr(Vec<Json>),
Obj(Vec<(String, Json)>),
}
/// Converter input options.
#[derive(Default, Clone)]
pub struct Options {
pub root_name: Option<String>,
}
/// Converter outcome. `error` is `None` when `ok` is true.
pub struct Outcome {
pub ok: bool,
pub code: String,
pub error: Option<String>,
}
/// A tiny recursive-descent JSON parser. Operates on a `Vec<char>` so byte
/// indexing and UTF-8 boundary concerns disappear at the cost of one up-front
/// allocation — a fine trade-off for a converter that runs once per call.
struct Parser {
chars: Vec<char>,
pos: usize,
}
impl Parser {
fn new(src: &str) -> Self {
Parser {
chars: src.chars().collect(),
pos: 0,
}
}
fn peek(&self) -> Option<char> {
self.chars.get(self.pos).copied()
}
/// Advance over JSON whitespace only — spec defines exactly four bytes,
/// so we don't use `char::is_whitespace` (which accepts Unicode space).
fn skip_ws(&mut self) {
while let Some(c) = self.peek() {
if matches!(c, ' ' | '\t' | '\n' | '\r') {
self.pos += 1;
} else {
break;
}
}
}
fn parse_value(&mut self) -> Result<Json, String> {
self.skip_ws();
match self.peek() {
None => Err("unexpected end of JSON input".into()),
Some('{') => self.parse_object(),
Some('[') => self.parse_array(),
Some('"') => self.parse_string().map(Json::Str),
Some('t') | Some('f') => self.parse_bool(),
Some('n') => self.parse_null(),
Some(c) if c == '-' || c.is_ascii_digit() => self.parse_number(),
Some(c) => Err(format!("unexpected character '{}'", c)),
}
}
fn parse_object(&mut self) -> Result<Json, String> {
self.pos += 1; // consume '{'
let mut entries: Vec<(String, Json)> = Vec::new();
self.skip_ws();
if self.peek() == Some('}') {
self.pos += 1;
return Ok(Json::Obj(entries));
}
loop {
self.skip_ws();
if self.peek() != Some('"') {
return Err("expected string key in object".into());
}
let key = self.parse_string()?;
self.skip_ws();
if self.peek() != Some(':') {
return Err("expected ':' after object key".into());
}
self.pos += 1; // consume ':'
let val = self.parse_value()?;
entries.push((key, val));
self.skip_ws();
match self.peek() {
Some(',') => {
self.pos += 1;
}
Some('}') => {
self.pos += 1;
break;
}
_ => return Err("expected ',' or '}' in object".into()),
}
}
Ok(Json::Obj(entries))
}
fn parse_array(&mut self) -> Result<Json, String> {
self.pos += 1; // consume '['
let mut items: Vec<Json> = Vec::new();
self.skip_ws();
if self.peek() == Some(']') {
self.pos += 1;
return Ok(Json::Arr(items));
}
loop {
let val = self.parse_value()?;
items.push(val);
self.skip_ws();
match self.peek() {
Some(',') => {
self.pos += 1;
}
Some(']') => {
self.pos += 1;
break;
}
_ => return Err("expected ',' or ']' in array".into()),
}
}
Ok(Json::Arr(items))
}
/// Parse a `"..."` token, resolving every standard escape sequence and
/// `\uXXXX` surrogate pairs into proper Rust `char`s.
fn parse_string(&mut self) -> Result<String, String> {
self.pos += 1; // consume opening quote
let mut out = String::new();
loop {
match self.chars.get(self.pos).copied() {
None => return Err("unterminated string".into()),
Some('"') => {
self.pos += 1;
return Ok(out);
}
Some('\\') => {
self.pos += 1;
let esc = self
.chars
.get(self.pos)
.copied()
.ok_or_else(|| "unterminated escape".to_string())?;
self.pos += 1;
match esc {
'"' => out.push('"'),
'\\' => out.push('\\'),
'/' => out.push('/'),
'b' => out.push('\u{0008}'),
'f' => out.push('\u{000C}'),
'n' => out.push('\n'),
'r' => out.push('\r'),
't' => out.push('\t'),
'u' => out.push(self.parse_unicode_escape()?),
other => return Err(format!("invalid escape \\{}", other)),
}
}
Some(c) => {
out.push(c);
self.pos += 1;
}
}
}
}
/// Read four hex digits; if it's a high surrogate, consume the trailing
/// low surrogate and combine into the proper astral-plane code point.
fn parse_unicode_escape(&mut self) -> Result<char, String> {
let code = self.read_hex4()?;
if (0xD800..=0xDBFF).contains(&code) {
// High surrogate: the JSON spec mandates a following low surrogate.
if self.chars.get(self.pos).copied() == Some('\\')
&& self.chars.get(self.pos + 1).copied() == Some('u')
{
self.pos += 2; // consume "\u"
let lo = self.read_hex4()?;
if !(0xDC00..=0xDFFF).contains(&lo) {
return Err("invalid low surrogate after high surrogate".into());
}
let combined = 0x10000 + ((code - 0xD800) << 10) + (lo - 0xDC00);
char::from_u32(combined).ok_or_else(|| "invalid surrogate pair".into())
} else {
Err("expected low surrogate after high surrogate".into())
}
} else {
char::from_u32(code).ok_or_else(|| "invalid unicode code point".into())
}
}
/// Read exactly four hexadecimal digits as a u32.
fn read_hex4(&mut self) -> Result<u32, String> {
let mut acc: u32 = 0;
for _ in 0..4 {
let h = self
.chars
.get(self.pos)
.copied()
.ok_or_else(|| "incomplete \\u escape".to_string())?;
self.pos += 1;
let d = h
.to_digit(16)
.ok_or_else(|| format!("invalid hex digit '{}'", h))?;
acc = acc * 16 + d;
}
Ok(acc)
}
fn parse_bool(&mut self) -> Result<Json, String> {
if self.match_lit("true") {
return Ok(Json::Bool(true));
}
if self.match_lit("false") {
return Ok(Json::Bool(false));
}
Err("invalid literal (expected true or false)".into())
}
fn parse_null(&mut self) -> Result<Json, String> {
if self.match_lit("null") {
return Ok(Json::Null);
}
Err("invalid literal (expected null)".into())
}
/// Match a fixed literal at the current position, advancing on success.
fn match_lit(&mut self, lit: &str) -> bool {
let lit_chars: Vec<char> = lit.chars().collect();
if self.pos + lit_chars.len() > self.chars.len() {
return false;
}
for (i, &c) in lit_chars.iter().enumerate() {
if self.chars[self.pos + i] != c {
return false;
}
}
self.pos += lit_chars.len();
true
}
/// Parse a JSON number. Stored as f64 to match JavaScript's number model
/// (lossy for integers beyond 2^53, just like JSON.parse).
fn parse_number(&mut self) -> Result<Json, String> {
let start = self.pos;
if self.peek() == Some('-') {
self.pos += 1;
}
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
if self.peek() == Some('.') {
self.pos += 1;
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
}
if matches!(self.peek(), Some('e') | Some('E')) {
self.pos += 1;
if matches!(self.peek(), Some('+') | Some('-')) {
self.pos += 1;
}
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
}
let text: String = self.chars[start..self.pos].iter().collect();
text.parse::<f64>()
.map(Json::Num)
.map_err(|_| format!("invalid number '{}'", text))
}
}
/// Reduce an arbitrary string to a usable JS identifier: drop every char
/// outside [A-Za-z0-9_$], replace each leading digit with '_', and default to
/// "schema" when nothing usable remains.
fn sanitize_var_name(name: &str) -> String {
let cleaned: String = name
.chars()
.filter(|c| c.is_ascii_alphanumeric() || *c == '_' || *c == '$')
.collect();
if cleaned.is_empty() {
return "schema".to_string();
}
let mut out = String::with_capacity(cleaned.len());
let mut leading = true;
for c in cleaned.chars() {
if leading && c.is_ascii_digit() {
out.push('_');
} else {
leading = false;
out.push(c);
}
}
out
}
/// Indent every non-empty line of `s` by `depth` spaces. Blank lines stay blank
/// so they don't pick up trailing whitespace.
fn pad(s: &str, depth: usize) -> String {
let prefix = " ".repeat(depth);
s.split('\n')
.map(|l| if l.is_empty() { l.to_string() } else { format!("{}{}", prefix, l) })
.collect::<Vec<_>>()
.join("\n")
}
/// Order-preserving de-duplication — the slice equivalent of JS
/// `[...new Set(seq)]`. A `HashSet` records seen items; the output `Vec` keeps
/// the original first-seen order.
fn distinct_vec(items: &[String]) -> Vec<String> {
let mut seen: HashMap<&str, ()> = HashMap::new();
let mut out: Vec<String> = Vec::new();
for s in items {
if !seen.contains_key(s.as_str()) {
seen.insert(s.as_str(), ());
out.push(s.clone());
}
}
out
}
/// Infer a Zod schema string for a parsed value at the given indentation depth.
fn infer_zod(value: &Json, indent: usize) -> String {
match value {
Json::Null => "z.null()".into(),
Json::Bool(_) => "z.boolean()".into(),
Json::Num(_) => "z.number()".into(),
Json::Str(_) => "z.string()".into(),
Json::Arr(items) => {
if items.is_empty() {
return "z.array(z.unknown())".into();
}
let types: Vec<String> =
items.iter().map(|e| infer_zod(e, indent + 2)).collect();
let distinct = distinct_vec(&types);
// Single shared element type → z.array(T). Multiple → z.union([...]).
// The union intentionally renders the *full* types list (with
// duplicates), matching the TypeScript reference exactly.
let inner = if distinct.len() == 1 {
distinct[0].clone()
} else {
format!(
"z.union([\n{}\n{}])",
pad(&types.join(",\n"), indent + 2),
pad("", indent),
)
};
format!("z.array({})", inner)
}
Json::Obj(entries) => {
if entries.is_empty() {
return "z.object({})".into();
}
let pad0 = " ".repeat(indent);
let pad1 = " ".repeat(indent + 2);
let fields: Vec<String> = entries
.iter()
.map(|(k, v)| format!("{}{}: {},", pad1, k, infer_zod(v, indent + 2)))
.collect();
format!("z.object({{\n{}\n{}}})", fields.join("\n"), pad0)
}
}
}
/// Convert a JSON string into a `const NAME = <zod schema>;` declaration.
/// Mirrors `JSON.parse`: exactly one value, with no trailing content.
pub fn json_to_zod(json_string: &str, opts: Options) -> Outcome {
let mut p = Parser::new(json_string);
let value = match p.parse_value() {
Ok(v) => v,
Err(e) => {
return Outcome {
ok: false,
code: String::new(),
error: Some(e),
}
}
};
// Reject trailing non-whitespace — JSON.parse does the same.
p.skip_ws();
if p.pos != p.chars.len() {
return Outcome {
ok: false,
code: String::new(),
error: Some("unexpected trailing characters in JSON input".into()),
};
}
let root = sanitize_var_name(opts.root_name.as_deref().unwrap_or("Root"));
Outcome {
ok: true,
code: format!("const {} = {};", root, infer_zod(&value, 2)),
error: None,
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →