Skip to content

JSON to Zod Schema — Rust source

Generate Zod validation schemas from JSON. Infers z.string, z.number, z.boolean, z.object, z.array, z.null, and z.union for mixed arrays.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! json-to-zod — Rust polyglot showcase port.
//!
//! Recursively infers a Zod schema string from a JSON value. Mixed-type arrays
//! collapse to z.union(...); plain objects become z.object({...}); empty arrays
//! and objects fall back to z.array(z.unknown()) / z.object({}). The converter
//! never panics — JSON parse failures and inference problems are returned as
//! `Outcome { ok: false, error: Some(...) }`.
//!
//! Ported from src/lib/jsonToZod.ts (CosmoDev).
//! Display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//!
//! Rust's standard library ships no JSON parser, so this file includes a small,
//! self-contained recursive-descent parser. Keeping it in-tree means the port
//! has zero external crates and full control over object key insertion order
//! (which a BTreeMap-based parser would sort away).

use std::collections::HashMap;

/// A JSON value. Object entries are stored in a `Vec<(String, Json)>` so the
/// parser preserves source insertion order — matching JavaScript's parser and
/// keeping generated field order stable.
#[derive(Debug, Clone)]
enum Json {
    Null,
    Bool(bool),
    Num(f64),
    Str(String),
    Arr(Vec<Json>),
    Obj(Vec<(String, Json)>),
}

/// Converter input options.
#[derive(Default, Clone)]
pub struct Options {
    pub root_name: Option<String>,
}

/// Converter outcome. `error` is `None` when `ok` is true.
pub struct Outcome {
    pub ok: bool,
    pub code: String,
    pub error: Option<String>,
}

/// A tiny recursive-descent JSON parser. Operates on a `Vec<char>` so byte
/// indexing and UTF-8 boundary concerns disappear at the cost of one up-front
/// allocation — a fine trade-off for a converter that runs once per call.
struct Parser {
    chars: Vec<char>,
    pos: usize,
}

impl Parser {
    fn new(src: &str) -> Self {
        Parser {
            chars: src.chars().collect(),
            pos: 0,
        }
    }

    fn peek(&self) -> Option<char> {
        self.chars.get(self.pos).copied()
    }

    /// Advance over JSON whitespace only — spec defines exactly four bytes,
    /// so we don't use `char::is_whitespace` (which accepts Unicode space).
    fn skip_ws(&mut self) {
        while let Some(c) = self.peek() {
            if matches!(c, ' ' | '\t' | '\n' | '\r') {
                self.pos += 1;
            } else {
                break;
            }
        }
    }

    fn parse_value(&mut self) -> Result<Json, String> {
        self.skip_ws();
        match self.peek() {
            None => Err("unexpected end of JSON input".into()),
            Some('{') => self.parse_object(),
            Some('[') => self.parse_array(),
            Some('"') => self.parse_string().map(Json::Str),
            Some('t') | Some('f') => self.parse_bool(),
            Some('n') => self.parse_null(),
            Some(c) if c == '-' || c.is_ascii_digit() => self.parse_number(),
            Some(c) => Err(format!("unexpected character '{}'", c)),
        }
    }

    fn parse_object(&mut self) -> Result<Json, String> {
        self.pos += 1; // consume '{'
        let mut entries: Vec<(String, Json)> = Vec::new();
        self.skip_ws();
        if self.peek() == Some('}') {
            self.pos += 1;
            return Ok(Json::Obj(entries));
        }
        loop {
            self.skip_ws();
            if self.peek() != Some('"') {
                return Err("expected string key in object".into());
            }
            let key = self.parse_string()?;
            self.skip_ws();
            if self.peek() != Some(':') {
                return Err("expected ':' after object key".into());
            }
            self.pos += 1; // consume ':'
            let val = self.parse_value()?;
            entries.push((key, val));
            self.skip_ws();
            match self.peek() {
                Some(',') => {
                    self.pos += 1;
                }
                Some('}') => {
                    self.pos += 1;
                    break;
                }
                _ => return Err("expected ',' or '}' in object".into()),
            }
        }
        Ok(Json::Obj(entries))
    }

    fn parse_array(&mut self) -> Result<Json, String> {
        self.pos += 1; // consume '['
        let mut items: Vec<Json> = Vec::new();
        self.skip_ws();
        if self.peek() == Some(']') {
            self.pos += 1;
            return Ok(Json::Arr(items));
        }
        loop {
            let val = self.parse_value()?;
            items.push(val);
            self.skip_ws();
            match self.peek() {
                Some(',') => {
                    self.pos += 1;
                }
                Some(']') => {
                    self.pos += 1;
                    break;
                }
                _ => return Err("expected ',' or ']' in array".into()),
            }
        }
        Ok(Json::Arr(items))
    }

    /// Parse a `"..."` token, resolving every standard escape sequence and
    /// `\uXXXX` surrogate pairs into proper Rust `char`s.
    fn parse_string(&mut self) -> Result<String, String> {
        self.pos += 1; // consume opening quote
        let mut out = String::new();
        loop {
            match self.chars.get(self.pos).copied() {
                None => return Err("unterminated string".into()),
                Some('"') => {
                    self.pos += 1;
                    return Ok(out);
                }
                Some('\\') => {
                    self.pos += 1;
                    let esc = self
                        .chars
                        .get(self.pos)
                        .copied()
                        .ok_or_else(|| "unterminated escape".to_string())?;
                    self.pos += 1;
                    match esc {
                        '"' => out.push('"'),
                        '\\' => out.push('\\'),
                        '/' => out.push('/'),
                        'b' => out.push('\u{0008}'),
                        'f' => out.push('\u{000C}'),
                        'n' => out.push('\n'),
                        'r' => out.push('\r'),
                        't' => out.push('\t'),
                        'u' => out.push(self.parse_unicode_escape()?),
                        other => return Err(format!("invalid escape \\{}", other)),
                    }
                }
                Some(c) => {
                    out.push(c);
                    self.pos += 1;
                }
            }
        }
    }

    /// Read four hex digits; if it's a high surrogate, consume the trailing
    /// low surrogate and combine into the proper astral-plane code point.
    fn parse_unicode_escape(&mut self) -> Result<char, String> {
        let code = self.read_hex4()?;
        if (0xD800..=0xDBFF).contains(&code) {
            // High surrogate: the JSON spec mandates a following low surrogate.
            if self.chars.get(self.pos).copied() == Some('\\')
                && self.chars.get(self.pos + 1).copied() == Some('u')
            {
                self.pos += 2; // consume "\u"
                let lo = self.read_hex4()?;
                if !(0xDC00..=0xDFFF).contains(&lo) {
                    return Err("invalid low surrogate after high surrogate".into());
                }
                let combined = 0x10000 + ((code - 0xD800) << 10) + (lo - 0xDC00);
                char::from_u32(combined).ok_or_else(|| "invalid surrogate pair".into())
            } else {
                Err("expected low surrogate after high surrogate".into())
            }
        } else {
            char::from_u32(code).ok_or_else(|| "invalid unicode code point".into())
        }
    }

    /// Read exactly four hexadecimal digits as a u32.
    fn read_hex4(&mut self) -> Result<u32, String> {
        let mut acc: u32 = 0;
        for _ in 0..4 {
            let h = self
                .chars
                .get(self.pos)
                .copied()
                .ok_or_else(|| "incomplete \\u escape".to_string())?;
            self.pos += 1;
            let d = h
                .to_digit(16)
                .ok_or_else(|| format!("invalid hex digit '{}'", h))?;
            acc = acc * 16 + d;
        }
        Ok(acc)
    }

    fn parse_bool(&mut self) -> Result<Json, String> {
        if self.match_lit("true") {
            return Ok(Json::Bool(true));
        }
        if self.match_lit("false") {
            return Ok(Json::Bool(false));
        }
        Err("invalid literal (expected true or false)".into())
    }

    fn parse_null(&mut self) -> Result<Json, String> {
        if self.match_lit("null") {
            return Ok(Json::Null);
        }
        Err("invalid literal (expected null)".into())
    }

    /// Match a fixed literal at the current position, advancing on success.
    fn match_lit(&mut self, lit: &str) -> bool {
        let lit_chars: Vec<char> = lit.chars().collect();
        if self.pos + lit_chars.len() > self.chars.len() {
            return false;
        }
        for (i, &c) in lit_chars.iter().enumerate() {
            if self.chars[self.pos + i] != c {
                return false;
            }
        }
        self.pos += lit_chars.len();
        true
    }

    /// Parse a JSON number. Stored as f64 to match JavaScript's number model
    /// (lossy for integers beyond 2^53, just like JSON.parse).
    fn parse_number(&mut self) -> Result<Json, String> {
        let start = self.pos;
        if self.peek() == Some('-') {
            self.pos += 1;
        }
        while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
            self.pos += 1;
        }
        if self.peek() == Some('.') {
            self.pos += 1;
            while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
                self.pos += 1;
            }
        }
        if matches!(self.peek(), Some('e') | Some('E')) {
            self.pos += 1;
            if matches!(self.peek(), Some('+') | Some('-')) {
                self.pos += 1;
            }
            while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
                self.pos += 1;
            }
        }
        let text: String = self.chars[start..self.pos].iter().collect();
        text.parse::<f64>()
            .map(Json::Num)
            .map_err(|_| format!("invalid number '{}'", text))
    }
}

/// Reduce an arbitrary string to a usable JS identifier: drop every char
/// outside [A-Za-z0-9_$], replace each leading digit with '_', and default to
/// "schema" when nothing usable remains.
fn sanitize_var_name(name: &str) -> String {
    let cleaned: String = name
        .chars()
        .filter(|c| c.is_ascii_alphanumeric() || *c == '_' || *c == '$')
        .collect();
    if cleaned.is_empty() {
        return "schema".to_string();
    }
    let mut out = String::with_capacity(cleaned.len());
    let mut leading = true;
    for c in cleaned.chars() {
        if leading && c.is_ascii_digit() {
            out.push('_');
        } else {
            leading = false;
            out.push(c);
        }
    }
    out
}

/// Indent every non-empty line of `s` by `depth` spaces. Blank lines stay blank
/// so they don't pick up trailing whitespace.
fn pad(s: &str, depth: usize) -> String {
    let prefix = " ".repeat(depth);
    s.split('\n')
        .map(|l| if l.is_empty() { l.to_string() } else { format!("{}{}", prefix, l) })
        .collect::<Vec<_>>()
        .join("\n")
}

/// Order-preserving de-duplication — the slice equivalent of JS
/// `[...new Set(seq)]`. A `HashSet` records seen items; the output `Vec` keeps
/// the original first-seen order.
fn distinct_vec(items: &[String]) -> Vec<String> {
    let mut seen: HashMap<&str, ()> = HashMap::new();
    let mut out: Vec<String> = Vec::new();
    for s in items {
        if !seen.contains_key(s.as_str()) {
            seen.insert(s.as_str(), ());
            out.push(s.clone());
        }
    }
    out
}

/// Infer a Zod schema string for a parsed value at the given indentation depth.
fn infer_zod(value: &Json, indent: usize) -> String {
    match value {
        Json::Null => "z.null()".into(),
        Json::Bool(_) => "z.boolean()".into(),
        Json::Num(_) => "z.number()".into(),
        Json::Str(_) => "z.string()".into(),
        Json::Arr(items) => {
            if items.is_empty() {
                return "z.array(z.unknown())".into();
            }
            let types: Vec<String> =
                items.iter().map(|e| infer_zod(e, indent + 2)).collect();
            let distinct = distinct_vec(&types);

            // Single shared element type → z.array(T). Multiple → z.union([...]).
            // The union intentionally renders the *full* types list (with
            // duplicates), matching the TypeScript reference exactly.
            let inner = if distinct.len() == 1 {
                distinct[0].clone()
            } else {
                format!(
                    "z.union([\n{}\n{}])",
                    pad(&types.join(",\n"), indent + 2),
                    pad("", indent),
                )
            };
            format!("z.array({})", inner)
        }
        Json::Obj(entries) => {
            if entries.is_empty() {
                return "z.object({})".into();
            }
            let pad0 = " ".repeat(indent);
            let pad1 = " ".repeat(indent + 2);
            let fields: Vec<String> = entries
                .iter()
                .map(|(k, v)| format!("{}{}: {},", pad1, k, infer_zod(v, indent + 2)))
                .collect();
            format!("z.object({{\n{}\n{}}})", fields.join("\n"), pad0)
        }
    }
}

/// Convert a JSON string into a `const NAME = <zod schema>;` declaration.
/// Mirrors `JSON.parse`: exactly one value, with no trailing content.
pub fn json_to_zod(json_string: &str, opts: Options) -> Outcome {
    let mut p = Parser::new(json_string);
    let value = match p.parse_value() {
        Ok(v) => v,
        Err(e) => {
            return Outcome {
                ok: false,
                code: String::new(),
                error: Some(e),
            }
        }
    };
    // Reject trailing non-whitespace — JSON.parse does the same.
    p.skip_ws();
    if p.pos != p.chars.len() {
        return Outcome {
            ok: false,
            code: String::new(),
            error: Some("unexpected trailing characters in JSON input".into()),
        };
    }

    let root = sanitize_var_name(opts.root_name.as_deref().unwrap_or("Root"));
    Outcome {
        ok: true,
        code: format!("const {} = {};", root, infer_zod(&value, 2)),
        error: None,
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →