JSON Validator — Rust source
Validate JSON and pinpoint errors with line and column numbers. Clear valid/invalid verdict plus the exact error location - runs entirely in your browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
// json-validator — Rust polyglot showcase port
// Language: Rust
// CosmoDev polyglot showcase port of the json-validator tool.
// Ported from src/lib/jsonValidate.ts (canonical TypeScript logic).
//
// This is display source — part of CosmoDev's polyglot tool pages (dev.cosmolabs.org).
//
// Pure JSON validation logic. The Rust standard library ships no JSON parser,
// so — unlike the TypeScript/JavaScript/Go/PHP/Python ports, which lean on a
// stdlib JSON engine — this file includes a small hand-written recursive-descent
// validator (std only, no serde and no external crates). On error it reports a
// precise byte offset, which we convert to a 1-based {line, column} just as the
// other ports convert the host engine's "at position N".
#![forbid(unsafe_code)]
use std::fmt;
/// A parse failure: the byte offset where the error was detected, plus a short
/// description. Distinct from stdlib error types so the public API leaks no
/// implementation detail.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ParseError {
pub offset: usize,
pub message: &'static str,
}
impl fmt::Display for ParseError {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(self.message)
}
}
impl std::error::Error for ParseError {}
/// The validation result. `error` and the location fields are `None` on success
/// or when the position cannot be determined — the Rust analogue of the
/// TypeScript `string | null` / `number | null`.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ValidationResult {
pub valid: bool,
pub error: Option<String>,
pub line: Option<usize>,
pub column: Option<usize>,
}
/// Extract the 0-based offset encoded as "at position N" in an engine error
/// message. The Rust parser carries the offset out-of-band on `ParseError`, so
/// this is kept for parity: it locates errors reported by *foreign* engines.
pub fn extract_offset(message: &str) -> Option<usize> {
const NEEDLE: &str = "at position ";
let tail = message.find(NEEDLE).map(|i| &message[i + NEEDLE.len()..])?;
let digits: String = tail.chars().take_while(|c| c.is_ascii_digit()).collect();
if digits.is_empty() {
return None;
}
digits.parse::<usize>().ok()
}
/// Coerce an error-like value into a display string. Rust is statically typed,
/// so TypeScript's dynamic `unknown` becomes a generic `Display` bound.
pub fn error_message<T: fmt::Display>(value: T) -> String {
format!("{value}")
}
/// Normalize an engine error message for display. Strips JavaScriptCore's
/// "JSON Parse error:" prefix (case-insensitive); a no-op for Rust messages.
pub fn normalize_error(message: &str) -> &str {
const PREFIX: &str = "JSON Parse error:";
// Byte-safe: PREFIX is pure ASCII, so its length is a char boundary.
let head = message.get(..PREFIX.len()).unwrap_or("");
if head.eq_ignore_ascii_case(PREFIX) {
message[PREFIX.len()..].trim_start()
} else {
message
}
}
/// Convert a 0-based byte offset into a 1-based (line, column).
/// Counts '\n', a lone '\r', and '\r\n' each as one line break. Columns count
/// Unicode scalar values, so a multibyte rune advances the column by one.
pub fn offset_to_line_col(text: &str, offset: usize) -> (usize, usize) {
let mut line = 1usize;
let mut column = 1usize;
let mut skip_lf = false; // set after a CR so a following LF folds into one break
for (i, ch) in text.char_indices() {
if i >= offset {
break;
}
if skip_lf && ch == '\n' {
skip_lf = false;
continue;
}
skip_lf = false;
match ch {
'\n' => {
line += 1;
column = 1;
}
'\r' => {
line += 1;
column = 1;
skip_lf = true;
}
_ => column += 1,
}
}
(line, column)
}
/// Map a parse error message to (line, column) (or (None, None)) using the input
/// text — locates errors reported by engines that embed "at position N".
pub fn locate_error(text: &str, message: &str) -> (Option<usize>, Option<usize>) {
match extract_offset(message) {
Some(offset) => {
let (line, column) = offset_to_line_col(text, offset);
(Some(line), Some(column))
}
None => (None, None),
}
}
/// Validate a JSON string. Never panics.
/// - empty / whitespace-only → invalid, "Input is empty", no location
/// - valid JSON → valid: true
/// - invalid JSON → valid: false, error message, and a 1-based
/// line/column derived from the byte offset
///
/// Rust's type system makes the TypeScript "non-string" guard unnecessary — the
/// parameter is already `&str`.
pub fn validate_json(text: &str) -> ValidationResult {
if text.trim().is_empty() {
return ValidationResult {
valid: false,
error: Some("Input is empty".to_string()),
line: None,
column: None,
};
}
match Parser::new(text).parse_document() {
Ok(()) => ValidationResult {
valid: true,
error: None,
line: None,
column: None,
},
Err(e) => {
let (line, column) = offset_to_line_col(text, e.offset);
ValidationResult {
valid: false,
error: Some(e.message.to_string()),
line: Some(line),
column: Some(column),
}
}
}
}
/// A minimal recursive-descent JSON validator. It walks the grammar without
/// building a value tree — we only care whether the input is well-formed and
/// where the first violation sits. The input is a valid `&str`, so UTF-8
/// sequences inside strings are already well-formed and need not be re-checked.
struct Parser<'a> {
bytes: &'a [u8],
pos: usize,
}
impl<'a> Parser<'a> {
fn new(text: &'a str) -> Self {
Self {
bytes: text.as_bytes(),
pos: 0,
}
}
/// Build an error at the current position.
fn err(&self, message: &'static str) -> ParseError {
ParseError {
offset: self.pos,
message,
}
}
fn peek(&self) -> Option<u8> {
self.bytes.get(self.pos).copied()
}
/// Advance one ASCII byte. Only called for ASCII consumption steps, so this
/// never splits a UTF-8 codepoint.
fn bump(&mut self) {
self.pos += 1;
}
/// Advance one full UTF-8 codepoint (used inside strings for multibyte
/// characters). The leading byte is consumed first, then any continuation
/// bytes (10xxxxxx).
fn bump_rune(&mut self) {
self.pos += 1; // leading byte
while let Some(b) = self.peek() {
if (b & 0xC0) == 0x80 {
self.pos += 1; // continuation byte
} else {
break;
}
}
}
/// Skip JSON whitespace: space, tab, line feed, carriage return.
fn skip_ws(&mut self) {
while matches!(self.peek(), Some(b' ' | b'\t' | b'\n' | b'\r')) {
self.pos += 1;
}
}
/// A JSON document: optional leading whitespace, exactly one value, then
/// optional trailing whitespace and end-of-input.
fn parse_document(&mut self) -> Result<(), ParseError> {
self.skip_ws();
self.parse_value()?;
self.skip_ws();
match self.peek() {
None => Ok(()),
Some(_) => Err(self.err("unexpected trailing characters")),
}
}
fn parse_value(&mut self) -> Result<(), ParseError> {
match self.peek() {
Some(b'{') => self.parse_object(),
Some(b'[') => self.parse_array(),
Some(b'"') => self.parse_string(),
Some(b't') | Some(b'f') => self.parse_bool(),
Some(b'n') => self.parse_null(),
Some(b'-') | Some(b'0'..=b'9') => self.parse_number(),
_ => Err(self.err("unexpected token")),
}
}
fn parse_object(&mut self) -> Result<(), ParseError> {
self.bump(); // consume '{'
self.skip_ws();
if self.peek() == Some(b'}') {
self.bump();
return Ok(());
}
loop {
self.skip_ws();
if self.peek() != Some(b'"') {
return Err(self.err("expected string key"));
}
self.parse_string()?;
self.skip_ws();
if self.peek() != Some(b':') {
return Err(self.err("expected ':' after key"));
}
self.bump(); // consume ':'
self.skip_ws();
self.parse_value()?;
self.skip_ws();
match self.peek() {
// Another member follows.
Some(b',') => self.bump(),
Some(b'}') => {
self.bump();
return Ok(());
}
_ => return Err(self.err("expected ',' or '}'")),
}
}
}
fn parse_array(&mut self) -> Result<(), ParseError> {
self.bump(); // consume '['
self.skip_ws();
if self.peek() == Some(b']') {
self.bump();
return Ok(());
}
loop {
self.skip_ws();
self.parse_value()?;
self.skip_ws();
match self.peek() {
// Another element follows.
Some(b',') => self.bump(),
Some(b']') => {
self.bump();
return Ok(());
}
_ => return Err(self.err("expected ',' or ']'")),
}
}
}
fn parse_string(&mut self) -> Result<(), ParseError> {
self.bump(); // opening '"'
loop {
match self.peek() {
None => return Err(self.err("unterminated string")),
Some(b'"') => {
self.bump();
return Ok(());
}
Some(b'\\') => {
self.bump();
self.parse_escape()?;
}
// Raw control characters (U+0000–U+001F) must be escaped in JSON.
Some(c) if c <= 0x1F => {
return Err(self.err("unescaped control character in string"));
}
Some(_) => self.bump_rune(),
}
}
}
/// Validate the bytes following a backslash.
fn parse_escape(&mut self) -> Result<(), ParseError> {
match self.peek() {
Some(b'"') | Some(b'\\') | Some(b'/') | Some(b'b') | Some(b'f') | Some(b'n')
| Some(b'r') | Some(b't') => {
self.bump();
Ok(())
}
Some(b'u') => {
self.bump(); // consume 'u'
let cp = self.parse_hex4()?;
self.check_surrogate(cp)
}
_ => Err(self.err("invalid escape sequence")),
}
}
/// Enforce correct UTF-16 surrogate pairing for '\u' escapes.
fn check_surrogate(&mut self, cp: u16) -> Result<(), ParseError> {
match cp {
0xD800..=0xDBFF => {
// High surrogate must be followed by '\u' + low surrogate.
if self.peek() != Some(b'\\') {
return Err(self.err("dangling high surrogate"));
}
self.bump();
if self.peek() != Some(b'u') {
return Err(self.err("expected '\\u' for surrogate pair"));
}
self.bump();
let lo = self.parse_hex4()?;
match lo {
0xDC00..=0xDFFF => Ok(()),
_ => Err(self.err("invalid low surrogate after high surrogate")),
}
}
0xDC00..=0xDFFF => Err(self.err("unexpected low surrogate")),
_ => Ok(()),
}
}
/// Read exactly four hexadecimal digits following a '\u'.
fn parse_hex4(&mut self) -> Result<u16, ParseError> {
let mut value: u16 = 0;
for _ in 0..4 {
let b = match self.peek() {
Some(b) => b,
None => return Err(self.err("incomplete '\\u' escape")),
};
let d = match b {
b'0'..=b'9' => b - b'0',
b'a'..=b'f' => b - b'a' + 10,
b'A'..=b'F' => b - b'A' + 10,
_ => return Err(self.err("invalid hex digit in '\\u' escape")),
};
value = value * 16 + d as u16;
self.bump();
}
Ok(value)
}
fn parse_number(&mut self) -> Result<(), ParseError> {
let start = self.pos;
if self.peek() == Some(b'-') {
self.bump();
}
// Integer part: '0' alone, or a non-zero digit followed by digits.
match self.peek() {
Some(b'0') => self.bump(),
Some(b'1'..=b'9') => {
self.bump();
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.bump();
}
}
_ => return Err(ParseError { offset: start, message: "invalid number" }),
}
// Optional fractional part.
if self.peek() == Some(b'.') {
self.bump();
if !matches!(self.peek(), Some(b'0'..=b'9')) {
return Err(self.err("expected digit after decimal point"));
}
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.bump();
}
}
// Optional exponent.
if matches!(self.peek(), Some(b'e') | Some(b'E')) {
self.bump();
if matches!(self.peek(), Some(b'+') | Some(b'-')) {
self.bump();
}
if !matches!(self.peek(), Some(b'0'..=b'9')) {
return Err(self.err("expected digit in exponent"));
}
while matches!(self.peek(), Some(b'0'..=b'9')) {
self.bump();
}
}
Ok(())
}
fn parse_bool(&mut self) -> Result<(), ParseError> {
if self.consume_keyword(b"true") || self.consume_keyword(b"false") {
Ok(())
} else {
Err(self.err("invalid literal"))
}
}
fn parse_null(&mut self) -> Result<(), ParseError> {
if self.consume_keyword(b"null") {
Ok(())
} else {
Err(self.err("invalid literal"))
}
}
/// Match a literal keyword at the current position; on success advance past it.
fn consume_keyword(&mut self, kw: &[u8]) -> bool {
if self.bytes.get(self.pos..self.pos + kw.len()) == Some(kw) {
self.pos += kw.len();
true
} else {
false
}
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →