JSON Repair — Rust source
Fix broken JSON - trailing commas, single quotes, unquoted keys, comments, Python constants, BOM and truncated documents - and get clean pretty-printed JSON plus a list of every repair applied. Runs entirely in your browser.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! json-repair — salvage broken JSON back to valid, pretty-printed JSON.
//!
//! Language: Rust (edition 2021, standard library only)
//! Source: CosmoDev polyglot showcase port of the JSON Repair tool, ported
//! from src/lib/jsonRepair.ts (the canonical TypeScript
//! implementation) and cli/json-repair/json-repair.go (the live Go
//! CLI twin).
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics (failures land in RepairResult::error).
//! - Functionally equivalent to the TS / Go references: same inputs -> same
//! outputs.
//! - Self-contained: std only (no crates.io dependencies — no `serde_json`).
//!
//! Approach: the repair passes walk `Vec<char>` code points exactly the way
//! the Go twin walks runes (TS indexes UTF-16 units — identical for the
//! ASCII/BMP content these passes reason about). The final parse-and-pretty-
//! print is a hand-rolled recursive-descent JSON validator that emits the
//! 2-space-indented shape `JSON.stringify(_, null, 2)` produces, while
//! copying strings and numbers VERBATIM (no float reformatting) and
//! preserving object key order — the same token-walk PrettyPrint the Go twin
//! uses, because Rust's std has no JSON.
//!
//! Divergences, documented: TS `/\s/` whitespace becomes `char::is_whitespace`
//! (same Unicode-whitespace spirit as the Go twin's `unicode.IsSpace`), and
//! TS `\b` word boundaries become explicit ASCII word-char neighbour checks
//! (the exact trick the Go twin uses). Lone `\uD800`-`\uDFFF` surrogate
//! escapes decode to U+FFFD instead of combining into a pair.
use std::fmt::Write as _;
/// Result of a repair run (mirrors the TS `RepairResult`).
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct RepairResult {
/// Pretty-printed JSON. Empty string when the input could not be repaired.
pub text: String,
/// Human-readable description of each repair pass that changed the text.
pub fixes: Vec<&'static str>,
/// True when at least one repair pass changed the text.
pub changed: bool,
/// True when the final text parses as JSON.
pub ok: bool,
/// Why the input could not be repaired (`ok == false`); `None` otherwise.
pub error: Option<String>,
}
type Span = (usize, usize); // [start, end] inclusive, quotes included
/// [start, end] index pairs of the double-quoted strings in text (quotes
/// included). An unterminated string extends to the last character.
pub fn string_spans(text: &str) -> Vec<Span> {
let r: Vec<char> = text.chars().collect();
let mut spans = Vec::new();
let mut open: Option<usize> = None;
let mut i = 0;
while i < r.len() {
let ch = r[i];
match open {
None => {
if ch == '"' {
open = Some(i);
}
}
Some(start) => {
if ch == '\\' {
i += 1; // skip the escaped character
} else if ch == '"' {
spans.push((start, i));
open = None;
}
}
}
i += 1;
}
if let Some(start) = open {
spans.push((start, r.len() - 1));
}
spans
}
/// True when index i sits inside one of the ordered, non-overlapping spans.
fn inside_span(spans: &[Span], i: usize) -> bool {
for &(start, end) in spans {
if i < start {
return false; // spans ascend: before this one means before all after it
}
if i <= end {
return true;
}
}
false
}
/// BOM, ZWSP, ZWNJ, ZWJ, word joiner — invisible characters that break parsers.
const INVISIBLE: [char; 5] = ['\u{feff}', '\u{200b}', '\u{200c}', '\u{200d}', '\u{2060}'];
/// Remove copy-paste invisible characters: BOM and zero-width joiners/spaces.
pub fn strip_invisible(text: &str) -> String {
text.chars().filter(|c| !INVISIBLE.contains(c)).collect()
}
fn is_space(ch: char) -> bool {
ch.is_whitespace()
}
fn is_key_start(ch: char) -> bool {
ch.is_ascii_alphabetic() || ch == '_' || ch == '$'
}
fn is_key_char(ch: char) -> bool {
is_key_start(ch) || ch.is_ascii_digit() || ch == '-'
}
fn is_word(ch: char) -> bool {
ch.is_ascii_alphanumeric() || ch == '_'
}
/// Strip `//` line comments and block comments, string-aware.
pub fn strip_comments(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let spans = string_spans(text);
let mut cuts: Vec<(usize, usize)> = Vec::new(); // [start, end exclusive)
let mut i = 0;
while i < r.len() {
if inside_span(&spans, i) {
i += 1;
continue;
}
if r[i] == '/' && i + 1 < r.len() && r[i + 1] == '/' {
let end = r[i..]
.iter()
.position(|&c| c == '\n')
.map(|p| i + p)
.unwrap_or(r.len());
cuts.push((i, end)); // keep the newline itself
i = end;
} else if r[i] == '/' && i + 1 < r.len() && r[i + 1] == '*' {
let end = find_seq(&r, "*/", i + 2).map(|p| p + 2).unwrap_or(r.len());
cuts.push((i, end));
i = end;
} else {
i += 1;
}
}
let mut out = String::new();
let mut prev = 0;
for (a, b) in cuts {
out.extend(r[prev..a].iter().copied());
prev = b;
}
out.extend(r[prev..].iter().copied());
out
}
/// Index of `seq` in `r` at or after `from`, byte-agnostic (char-wise).
fn find_seq(r: &[char], seq: &str, from: usize) -> Option<usize> {
let s: Vec<char> = seq.chars().collect();
if s.is_empty() {
return None;
}
let mut i = from;
while i + s.len() <= r.len() {
if r[i..i + s.len()] == s[..] {
return Some(i);
}
i += 1;
}
None
}
/// Convert single-quoted strings/keys to double-quoted JSON strings:
/// escape inner double quotes, collapse `\'` to `'`, keep every other escape.
pub fn single_to_double_quotes(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let mut out = String::new();
let mut i = 0;
while i < r.len() {
let ch = r[i];
if ch == '"' {
// Copy a double-quoted string verbatim (apostrophes stay put).
let mut j = i + 1;
while j < r.len() {
if r[j] == '\\' {
j += 2;
} else if r[j] == '"' {
j += 1;
break;
} else {
j += 1;
}
}
let j = j.min(r.len());
out.extend(r[i..j].iter().copied());
i = j;
} else if ch == '\'' {
let mut body = String::new();
let mut j = i + 1;
while j < r.len() {
let c = r[j];
if c == '\\' && j + 1 < r.len() {
let next = r[j + 1];
if next == '\'' {
body.push('\'');
} else {
body.push(c);
body.push(next);
}
j += 2;
} else if c == '\'' {
j += 1;
break;
} else if c == '"' {
body.push_str("\\\"");
j += 1;
} else {
body.push(c);
j += 1;
}
}
out.push('"');
out.push_str(&body);
out.push('"');
i = j;
} else {
out.push(ch);
i += 1;
}
}
out
}
/// Wrap bare identifier keys (`{name: 1}` -> `{"name": 1}`), string-aware.
pub fn quote_unquoted_keys(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let spans = string_spans(text);
let mut edits: Vec<(usize, usize)> = Vec::new();
let mut i = 0;
while i < r.len() {
if inside_span(&spans, i) {
i += 1;
continue;
}
let ch = r[i];
if ch != '{' && ch != ',' {
i += 1;
continue;
}
let mut j = i + 1;
while j < r.len() && is_space(r[j]) {
j += 1;
}
if j >= r.len() || !is_key_start(r[j]) {
i += 1;
continue;
}
let mut k = j;
while k < r.len() && is_key_char(r[k]) {
k += 1;
}
let mut l = k;
while l < r.len() && is_space(r[l]) {
l += 1;
}
if l < r.len() && r[l] == ':' {
edits.push((j, k));
}
i += 1;
}
let mut out = String::new();
let mut prev = 0;
for (a, b) in edits {
out.extend(r[prev..a].iter().copied());
out.push('"');
out.extend(r[a..b].iter().copied());
out.push('"');
prev = b;
}
out.extend(r[prev..].iter().copied());
out
}
const PY_CONSTANTS: [(&str, &str); 3] = [("True", "true"), ("False", "false"), ("None", "null")];
/// Rewrite bare Python constants (True/False/None) to JSON
/// (true/false/null), string-aware. Word boundaries are explicit ASCII
/// word-char neighbour checks — the same trick the Go twin uses for `\b`.
pub fn fix_python_constants(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let spans = string_spans(text);
let words: Vec<(Vec<char>, &str)> = PY_CONSTANTS
.iter()
.map(|(w, rep)| (w.chars().collect(), *rep))
.collect();
let mut edits: Vec<(usize, usize, &str)> = Vec::new();
let mut i = 0;
while i < r.len() {
for (w, rep) in &words {
let len = w.len();
if i + len > r.len() || r[i..i + len] != w[..] {
continue;
}
let before = i == 0 || !is_word(r[i - 1]);
let after = i + len >= r.len() || !is_word(r[i + len]);
if before && after && !inside_span(&spans, i) {
edits.push((i, i + len, rep));
}
}
i += 1;
}
let mut out = String::new();
let mut prev = 0;
for (a, b, rep) in edits {
out.extend(r[prev..a].iter().copied());
out.push_str(rep);
prev = b;
}
out.extend(r[prev..].iter().copied());
out
}
/// Remove commas followed only by whitespace and a closing `}` or `]`,
/// string-aware.
pub fn strip_trailing_commas(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let spans = string_spans(text);
let mut cuts: Vec<usize> = Vec::new();
let mut i = 0;
while i < r.len() {
if r[i] == ',' && !inside_span(&spans, i) {
let mut j = i + 1;
while j < r.len() && is_space(r[j]) {
j += 1;
}
if j < r.len() && (r[j] == '}' || r[j] == ']') {
cuts.push(i);
}
}
i += 1;
}
let mut out = String::new();
let mut prev = 0;
for c in cuts {
out.extend(r[prev..c].iter().copied());
prev = c + 1;
}
out.extend(r[prev..].iter().copied());
out
}
/// Recover truncated JSON: close an unterminated string, drop a dangling
/// comma, give a dangling colon a null value, then close every still-open
/// bracket in reverse order.
pub fn close_truncated(text: &str) -> String {
let r: Vec<char> = text.chars().collect();
let mut in_string = false;
let mut stack: Vec<char> = Vec::new();
let mut i = 0;
while i < r.len() {
let ch = r[i];
if in_string {
if ch == '\\' {
i += 1;
} else if ch == '"' {
in_string = false;
}
i += 1;
continue;
}
match ch {
'"' => in_string = true,
'{' | '[' => stack.push(ch),
'}' | ']' => {
stack.pop();
}
_ => {}
}
i += 1;
}
let mut out = String::from(text);
if in_string {
out.push('"');
}
while out.ends_with(|c: char| c.is_whitespace() || c == ',') {
out.pop();
}
if out.ends_with(':') {
out.push_str(" null");
}
let mut closers = String::new();
for c in stack.iter().rev() {
closers.push(if *c == '{' { '}' } else { ']' });
}
out.push_str(&closers);
out
}
/// The pass pipeline: (label, pass); each applied only while it changes the text.
const PASSES: [(&str, fn(&str) -> String); 7] = [
("Removed invisible characters (BOM / zero-width)", strip_invisible),
("Converted single quotes to double quotes", single_to_double_quotes),
("Stripped JavaScript comments", strip_comments),
("Quoted unquoted keys", quote_unquoted_keys),
("Converted Python constants (True/False/None)", fix_python_constants),
("Removed trailing commas", strip_trailing_commas),
("Closed truncated brackets", close_truncated),
];
/// Repair broken JSON and pretty-print the result. Never panics.
/// - empty / whitespace-only input -> ok=false, "Input is empty"
/// - input that already parses -> same shape, changed=false, no fixes
/// - repairable input -> pretty text + a description per pass that fired
/// - unrepairable input -> ok=false, the parser's error message
pub fn repair(text: &str) -> RepairResult {
if text.trim().is_empty() {
return RepairResult {
error: Some("Input is empty".into()),
..RepairResult::default()
};
}
// Already clean: pretty-print and say so.
if let Ok(pretty) = pretty_print(text) {
return RepairResult {
text: pretty,
ok: true,
..RepairResult::default()
};
}
let mut work = text.to_string();
let mut fixes: Vec<&'static str> = Vec::new();
for (label, pass) in PASSES {
let next = pass(&work);
if next != work {
fixes.push(label);
work = next;
}
}
let changed = !fixes.is_empty();
match pretty_print(&work) {
Ok(pretty) => RepairResult {
text: pretty,
fixes,
changed,
ok: true,
error: None,
},
Err(err) => RepairResult {
fixes,
changed,
ok: false,
error: Some(err),
..RepairResult::default()
},
}
}
/// Parse JSON and re-emit it pretty-printed with a 2-space indent (the shape
/// `JSON.stringify(_, null, 2)` produces). Strings and numbers are copied
/// verbatim and object key order is preserved. Err(String) on any invalid
/// input — the contract of the TS `prettyPrint`'s thrown error.
pub fn pretty_print(text: &str) -> Result<String, String> {
let r: Vec<char> = text.chars().collect();
let mut p = Parser { r: &r, pos: 0 };
p.skip_ws();
let mut out = String::new();
p.write_value(&mut out, 0)?;
p.skip_ws();
if p.pos != p.r.len() {
return Err("json: unexpected trailing data".into());
}
Ok(out)
}
/// Minimal recursive-descent JSON parser that writes the pretty-printed form
/// as it walks (RFC 8259 grammar, `json`-style error strings).
struct Parser<'a> {
r: &'a [char],
pos: usize,
}
impl<'a> Parser<'a> {
fn peek(&self) -> Option<char> {
self.r.get(self.pos).copied()
}
fn skip_ws(&mut self) {
while matches!(self.peek(), Some(' ' | '\t' | '\n' | '\r')) {
self.pos += 1;
}
}
fn write_value(&mut self, out: &mut String, depth: usize) -> Result<(), String> {
match self.peek() {
Some('{') => self.write_object(out, depth),
Some('[') => self.write_array(out, depth),
Some('"') => {
let s = self.parse_string()?;
out.push_str(&s); // already the re-encoded quoted form
Ok(())
}
Some('t') => self.expect_literal("true", out),
Some('f') => self.expect_literal("false", out),
Some('n') => self.expect_literal("null", out),
Some(c) if c == '-' || c.is_ascii_digit() => {
let n = self.parse_number()?;
out.push_str(&n);
Ok(())
}
_ => Err("json: invalid value".into()),
}
}
fn expect_literal(&mut self, lit: &str, out: &mut String) -> Result<(), String> {
for ch in lit.chars() {
if self.peek() != Some(ch) {
return Err(format!("json: expected {lit}"));
}
self.pos += 1;
}
out.push_str(lit);
Ok(())
}
fn write_object(&mut self, out: &mut String, depth: usize) -> Result<(), String> {
self.pos += 1; // consume '{'
let indent = " ".repeat(depth);
let inner = " ".repeat(depth + 1);
self.skip_ws();
if self.peek() == Some('}') {
self.pos += 1;
out.push_str("{}"); // empty object, like JSON.stringify
return Ok(());
}
out.push_str("{\n");
loop {
self.skip_ws();
let key = self.parse_string()?;
self.skip_ws();
if self.peek() != Some(':') {
return Err("json: missing ':' after object key".into());
}
self.pos += 1;
out.push_str(&inner);
out.push_str(&key);
out.push_str(": ");
self.skip_ws();
self.write_value(out, depth + 1)?;
self.skip_ws();
match self.peek() {
Some(',') => {
self.pos += 1;
out.push_str(",\n");
}
Some('}') => {
self.pos += 1;
let _ = write!(out, "\n{indent}}}");
return Ok(());
}
_ => return Err("json: expected ',' or '}' in object".into()),
}
}
}
fn write_array(&mut self, out: &mut String, depth: usize) -> Result<(), String> {
self.pos += 1; // consume '['
let indent = " ".repeat(depth);
let inner = " ".repeat(depth + 1);
self.skip_ws();
if self.peek() == Some(']') {
self.pos += 1;
out.push_str("[]"); // empty array, like JSON.stringify
return Ok(());
}
out.push_str("[\n");
loop {
self.skip_ws();
out.push_str(&inner);
self.write_value(out, depth + 1)?;
self.skip_ws();
match self.peek() {
Some(',') => {
self.pos += 1;
out.push_str(",\n");
}
Some(']') => {
self.pos += 1;
let _ = write!(out, "\n{indent}]");
return Ok(());
}
_ => return Err("json: expected ',' or ']' in array".into()),
}
}
}
/// Parse a double-quoted string, returning it RE-ENCODED with exactly the
/// escapes JSON.stringify produces (the quote, the backslash, control
/// characters; everything else — including non-ASCII — stays raw).
/// Rejects invalid escapes and raw control characters.
fn parse_string(&mut self) -> Result<String, String> {
if self.peek() != Some('"') {
return Err("json: expected string".into());
}
self.pos += 1;
let mut value = String::new();
loop {
let ch = self.peek().ok_or("json: unterminated string")?;
self.pos += 1;
match ch {
'"' => break,
'\\' => {
let esc = self.peek().ok_or("json: unterminated escape")?;
self.pos += 1;
match esc {
'"' => value.push('"'),
'\\' => value.push('\\'),
'/' => value.push('/'),
'b' => value.push('\u{0008}'),
'f' => value.push('\u{000c}'),
'n' => value.push('\n'),
'r' => value.push('\r'),
't' => value.push('\t'),
'u' => {
let mut hex: u32 = 0;
for _ in 0..4 {
let h = self.peek().ok_or("json: incomplete \\u escape")?;
self.pos += 1;
let d = h.to_digit(16).ok_or("json: invalid \\u escape")?;
hex = hex * 16 + d;
}
value.push(char::from_u32(hex).unwrap_or('\u{fffd}'));
}
_ => return Err("json: invalid escape".into()),
}
}
c if (c as u32) < 0x20 => return Err("json: control character in string".into()),
c => value.push(c),
}
}
let mut out = String::from("\"");
for ch in value.chars() {
match ch {
'"' => out.push_str("\\\""),
'\\' => out.push_str("\\\\"),
'\u{0008}' => out.push_str("\\b"),
'\u{000c}' => out.push_str("\\f"),
'\n' => out.push_str("\\n"),
'\r' => out.push_str("\\r"),
'\t' => out.push_str("\\t"),
c if (c as u32) < 0x20 => {
let _ = write!(out, "\\u{:04x}", c as u32);
}
c => out.push(c),
}
}
out.push('"');
Ok(out)
}
/// Scan a JSON number (RFC 8259 grammar) and copy it verbatim.
fn parse_number(&mut self) -> Result<String, String> {
let start = self.pos;
if self.peek() == Some('-') {
self.pos += 1;
}
// int part: 0 | [1-9][0-9]*
match self.peek() {
Some('0') => self.pos += 1,
Some(c) if c.is_ascii_digit() => {
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
}
_ => return Err("json: invalid number".into()),
}
if self.peek() == Some('.') {
self.pos += 1;
if !matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
return Err("json: invalid number frac".into());
}
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
}
if matches!(self.peek(), Some('e' | 'E')) {
self.pos += 1;
if matches!(self.peek(), Some('+' | '-')) {
self.pos += 1;
}
if !matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
return Err("json: invalid number exp".into());
}
while matches!(self.peek(), Some(c) if c.is_ascii_digit()) {
self.pos += 1;
}
}
Ok(self.r[start..self.pos].iter().collect())
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →