Skip to content

JSON Validator — Rust source

Validate JSON and pinpoint errors with line and column numbers. Clear valid/invalid verdict plus the exact error location - runs entirely in your browser.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

// json-validator — Rust polyglot showcase port
// Language: Rust
// CosmoDev polyglot showcase port of the json-validator tool.
// Ported from src/lib/jsonValidate.ts (canonical TypeScript logic).
//
// This is display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//
// Pure JSON validation logic. The Rust standard library ships no JSON parser,
// so — unlike the TypeScript/JavaScript/Go/PHP/Python ports, which lean on a
// stdlib JSON engine — this file includes a small hand-written recursive-descent
// validator (std only, no serde and no external crates). On error it reports a
// precise byte offset, which we convert to a 1-based {line, column} just as the
// other ports convert the host engine's "at position N".

#![forbid(unsafe_code)]

use std::fmt;

/// A parse failure: the byte offset where the error was detected, plus a short
/// description. Distinct from stdlib error types so the public API leaks no
/// implementation detail.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ParseError {
    pub offset: usize,
    pub message: &'static str,
}

impl fmt::Display for ParseError {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        f.write_str(self.message)
    }
}

impl std::error::Error for ParseError {}

/// The validation result. `error` and the location fields are `None` on success
/// or when the position cannot be determined — the Rust analogue of the
/// TypeScript `string | null` / `number | null`.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ValidationResult {
    pub valid: bool,
    pub error: Option<String>,
    pub line: Option<usize>,
    pub column: Option<usize>,
}

/// Extract the 0-based offset encoded as "at position N" in an engine error
/// message. The Rust parser carries the offset out-of-band on `ParseError`, so
/// this is kept for parity: it locates errors reported by *foreign* engines.
pub fn extract_offset(message: &str) -> Option<usize> {
    const NEEDLE: &str = "at position ";
    let tail = message.find(NEEDLE).map(|i| &message[i + NEEDLE.len()..])?;
    let digits: String = tail.chars().take_while(|c| c.is_ascii_digit()).collect();
    if digits.is_empty() {
        return None;
    }
    digits.parse::<usize>().ok()
}

/// Coerce an error-like value into a display string. Rust is statically typed,
/// so TypeScript's dynamic `unknown` becomes a generic `Display` bound.
pub fn error_message<T: fmt::Display>(value: T) -> String {
    format!("{value}")
}

/// Normalize an engine error message for display. Strips JavaScriptCore's
/// "JSON Parse error:" prefix (case-insensitive); a no-op for Rust messages.
pub fn normalize_error(message: &str) -> &str {
    const PREFIX: &str = "JSON Parse error:";
    // Byte-safe: PREFIX is pure ASCII, so its length is a char boundary.
    let head = message.get(..PREFIX.len()).unwrap_or("");
    if head.eq_ignore_ascii_case(PREFIX) {
        message[PREFIX.len()..].trim_start()
    } else {
        message
    }
}

/// Convert a 0-based byte offset into a 1-based (line, column).
/// Counts '\n', a lone '\r', and '\r\n' each as one line break. Columns count
/// Unicode scalar values, so a multibyte rune advances the column by one.
pub fn offset_to_line_col(text: &str, offset: usize) -> (usize, usize) {
    let mut line = 1usize;
    let mut column = 1usize;
    let mut skip_lf = false; // set after a CR so a following LF folds into one break
    for (i, ch) in text.char_indices() {
        if i >= offset {
            break;
        }
        if skip_lf && ch == '\n' {
            skip_lf = false;
            continue;
        }
        skip_lf = false;
        match ch {
            '\n' => {
                line += 1;
                column = 1;
            }
            '\r' => {
                line += 1;
                column = 1;
                skip_lf = true;
            }
            _ => column += 1,
        }
    }
    (line, column)
}

/// Map a parse error message to (line, column) (or (None, None)) using the input
/// text — locates errors reported by engines that embed "at position N".
pub fn locate_error(text: &str, message: &str) -> (Option<usize>, Option<usize>) {
    match extract_offset(message) {
        Some(offset) => {
            let (line, column) = offset_to_line_col(text, offset);
            (Some(line), Some(column))
        }
        None => (None, None),
    }
}

/// Validate a JSON string. Never panics.
///  - empty / whitespace-only → invalid, "Input is empty", no location
///  - valid JSON              → valid: true
///  - invalid JSON            → valid: false, error message, and a 1-based
///                              line/column derived from the byte offset
///
/// Rust's type system makes the TypeScript "non-string" guard unnecessary — the
/// parameter is already `&str`.
pub fn validate_json(text: &str) -> ValidationResult {
    if text.trim().is_empty() {
        return ValidationResult {
            valid: false,
            error: Some("Input is empty".to_string()),
            line: None,
            column: None,
        };
    }
    match Parser::new(text).parse_document() {
        Ok(()) => ValidationResult {
            valid: true,
            error: None,
            line: None,
            column: None,
        },
        Err(e) => {
            let (line, column) = offset_to_line_col(text, e.offset);
            ValidationResult {
                valid: false,
                error: Some(e.message.to_string()),
                line: Some(line),
                column: Some(column),
            }
        }
    }
}

/// A minimal recursive-descent JSON validator. It walks the grammar without
/// building a value tree — we only care whether the input is well-formed and
/// where the first violation sits. The input is a valid `&str`, so UTF-8
/// sequences inside strings are already well-formed and need not be re-checked.
struct Parser<'a> {
    bytes: &'a [u8],
    pos: usize,
}

impl<'a> Parser<'a> {
    fn new(text: &'a str) -> Self {
        Self {
            bytes: text.as_bytes(),
            pos: 0,
        }
    }

    /// Build an error at the current position.
    fn err(&self, message: &'static str) -> ParseError {
        ParseError {
            offset: self.pos,
            message,
        }
    }

    fn peek(&self) -> Option<u8> {
        self.bytes.get(self.pos).copied()
    }

    /// Advance one ASCII byte. Only called for ASCII consumption steps, so this
    /// never splits a UTF-8 codepoint.
    fn bump(&mut self) {
        self.pos += 1;
    }

    /// Advance one full UTF-8 codepoint (used inside strings for multibyte
    /// characters). The leading byte is consumed first, then any continuation
    /// bytes (10xxxxxx).
    fn bump_rune(&mut self) {
        self.pos += 1; // leading byte
        while let Some(b) = self.peek() {
            if (b & 0xC0) == 0x80 {
                self.pos += 1; // continuation byte
            } else {
                break;
            }
        }
    }

    /// Skip JSON whitespace: space, tab, line feed, carriage return.
    fn skip_ws(&mut self) {
        while matches!(self.peek(), Some(b' ' | b'\t' | b'\n' | b'\r')) {
            self.pos += 1;
        }
    }

    /// A JSON document: optional leading whitespace, exactly one value, then
    /// optional trailing whitespace and end-of-input.
    fn parse_document(&mut self) -> Result<(), ParseError> {
        self.skip_ws();
        self.parse_value()?;
        self.skip_ws();
        match self.peek() {
            None => Ok(()),
            Some(_) => Err(self.err("unexpected trailing characters")),
        }
    }

    fn parse_value(&mut self) -> Result<(), ParseError> {
        match self.peek() {
            Some(b'{') => self.parse_object(),
            Some(b'[') => self.parse_array(),
            Some(b'"') => self.parse_string(),
            Some(b't') | Some(b'f') => self.parse_bool(),
            Some(b'n') => self.parse_null(),
            Some(b'-') | Some(b'0'..=b'9') => self.parse_number(),
            _ => Err(self.err("unexpected token")),
        }
    }

    fn parse_object(&mut self) -> Result<(), ParseError> {
        self.bump(); // consume '{'
        self.skip_ws();
        if self.peek() == Some(b'}') {
            self.bump();
            return Ok(());
        }
        loop {
            self.skip_ws();
            if self.peek() != Some(b'"') {
                return Err(self.err("expected string key"));
            }
            self.parse_string()?;
            self.skip_ws();
            if self.peek() != Some(b':') {
                return Err(self.err("expected ':' after key"));
            }
            self.bump(); // consume ':'
            self.skip_ws();
            self.parse_value()?;
            self.skip_ws();
            match self.peek() {
                // Another member follows.
                Some(b',') => self.bump(),
                Some(b'}') => {
                    self.bump();
                    return Ok(());
                }
                _ => return Err(self.err("expected ',' or '}'")),
            }
        }
    }

    fn parse_array(&mut self) -> Result<(), ParseError> {
        self.bump(); // consume '['
        self.skip_ws();
        if self.peek() == Some(b']') {
            self.bump();
            return Ok(());
        }
        loop {
            self.skip_ws();
            self.parse_value()?;
            self.skip_ws();
            match self.peek() {
                // Another element follows.
                Some(b',') => self.bump(),
                Some(b']') => {
                    self.bump();
                    return Ok(());
                }
                _ => return Err(self.err("expected ',' or ']'")),
            }
        }
    }

    fn parse_string(&mut self) -> Result<(), ParseError> {
        self.bump(); // opening '"'
        loop {
            match self.peek() {
                None => return Err(self.err("unterminated string")),
                Some(b'"') => {
                    self.bump();
                    return Ok(());
                }
                Some(b'\\') => {
                    self.bump();
                    self.parse_escape()?;
                }
                // Raw control characters (U+0000–U+001F) must be escaped in JSON.
                Some(c) if c <= 0x1F => {
                    return Err(self.err("unescaped control character in string"));
                }
                Some(_) => self.bump_rune(),
            }
        }
    }

    /// Validate the bytes following a backslash.
    fn parse_escape(&mut self) -> Result<(), ParseError> {
        match self.peek() {
            Some(b'"') | Some(b'\\') | Some(b'/') | Some(b'b') | Some(b'f') | Some(b'n')
            | Some(b'r') | Some(b't') => {
                self.bump();
                Ok(())
            }
            Some(b'u') => {
                self.bump(); // consume 'u'
                let cp = self.parse_hex4()?;
                self.check_surrogate(cp)
            }
            _ => Err(self.err("invalid escape sequence")),
        }
    }

    /// Enforce correct UTF-16 surrogate pairing for '\u' escapes.
    fn check_surrogate(&mut self, cp: u16) -> Result<(), ParseError> {
        match cp {
            0xD800..=0xDBFF => {
                // High surrogate must be followed by '\u' + low surrogate.
                if self.peek() != Some(b'\\') {
                    return Err(self.err("dangling high surrogate"));
                }
                self.bump();
                if self.peek() != Some(b'u') {
                    return Err(self.err("expected '\\u' for surrogate pair"));
                }
                self.bump();
                let lo = self.parse_hex4()?;
                match lo {
                    0xDC00..=0xDFFF => Ok(()),
                    _ => Err(self.err("invalid low surrogate after high surrogate")),
                }
            }
            0xDC00..=0xDFFF => Err(self.err("unexpected low surrogate")),
            _ => Ok(()),
        }
    }

    /// Read exactly four hexadecimal digits following a '\u'.
    fn parse_hex4(&mut self) -> Result<u16, ParseError> {
        let mut value: u16 = 0;
        for _ in 0..4 {
            let b = match self.peek() {
                Some(b) => b,
                None => return Err(self.err("incomplete '\\u' escape")),
            };
            let d = match b {
                b'0'..=b'9' => b - b'0',
                b'a'..=b'f' => b - b'a' + 10,
                b'A'..=b'F' => b - b'A' + 10,
                _ => return Err(self.err("invalid hex digit in '\\u' escape")),
            };
            value = value * 16 + d as u16;
            self.bump();
        }
        Ok(value)
    }

    fn parse_number(&mut self) -> Result<(), ParseError> {
        let start = self.pos;
        if self.peek() == Some(b'-') {
            self.bump();
        }
        // Integer part: '0' alone, or a non-zero digit followed by digits.
        match self.peek() {
            Some(b'0') => self.bump(),
            Some(b'1'..=b'9') => {
                self.bump();
                while matches!(self.peek(), Some(b'0'..=b'9')) {
                    self.bump();
                }
            }
            _ => return Err(ParseError { offset: start, message: "invalid number" }),
        }
        // Optional fractional part.
        if self.peek() == Some(b'.') {
            self.bump();
            if !matches!(self.peek(), Some(b'0'..=b'9')) {
                return Err(self.err("expected digit after decimal point"));
            }
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.bump();
            }
        }
        // Optional exponent.
        if matches!(self.peek(), Some(b'e') | Some(b'E')) {
            self.bump();
            if matches!(self.peek(), Some(b'+') | Some(b'-')) {
                self.bump();
            }
            if !matches!(self.peek(), Some(b'0'..=b'9')) {
                return Err(self.err("expected digit in exponent"));
            }
            while matches!(self.peek(), Some(b'0'..=b'9')) {
                self.bump();
            }
        }
        Ok(())
    }

    fn parse_bool(&mut self) -> Result<(), ParseError> {
        if self.consume_keyword(b"true") || self.consume_keyword(b"false") {
            Ok(())
        } else {
            Err(self.err("invalid literal"))
        }
    }

    fn parse_null(&mut self) -> Result<(), ParseError> {
        if self.consume_keyword(b"null") {
            Ok(())
        } else {
            Err(self.err("invalid literal"))
        }
    }

    /// Match a literal keyword at the current position; on success advance past it.
    fn consume_keyword(&mut self, kw: &[u8]) -> bool {
        if self.bytes.get(self.pos..self.pos + kw.len()) == Some(kw) {
            self.pos += kw.len();
            true
        } else {
            false
        }
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →