Regex Explainer — Rust source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! Regex Explainer — Rust port.
//!
//! Language: Rust (std-only — no external crates).
//! Source: CosmoDev polyglot showcase port of the `regex-explainer` tool.
//! Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
//!
//! What it does:
//! Tokenizes a regular expression into labeled tokens — anchors, escapes,
//! character classes, quantifiers, groups, alternation, and literals — and
//! describes each flag. The explainer never panics: an invalid pattern yields
//! a structured error result.
//!
//! Engine note: Rust's standard library has no regular-expression engine, and
//! pulling in the external `regex` crate would violate the showcase's
//! dependency-free contract. The TS original validates by asking the JS regex
//! engine to compile the pattern; here we instead perform a lightweight
//! *structural* validation (balanced `[...]` and `(...)`, no trailing `\`).
//! This catches the common malformed-input cases without false negatives on
//! the syntactic constructs this tokenizer recognizes (lookaround,
//! backreferences, etc., which a real engine might still reject).
//!
//! This is display source — part of CosmoDev's polyglot tool pages.
/// A labeled slice of a regex pattern.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct RegexToken {
pub token: String,
pub description: String,
}
/// A flag letter paired with its human description.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct FlagInfo {
pub flag: String,
pub description: String,
}
/// The full output of [`explain_regex`].
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct ExplainResult {
pub ok: bool,
pub tokens: Vec<RegexToken>,
pub flags: Vec<FlagInfo>,
pub error: Option<String>,
}
/// Human description for a single flag letter, or `None` if unrecognized.
/// Mirrors the TS `describeFlag` (returns `string | null`).
pub fn describe_flag(flag: &str) -> Option<&'static str> {
match flag {
"g" => Some("global — find all matches"),
"i" => Some("case-insensitive"),
"m" => Some("multiline (^ and $ match line boundaries)"),
"s" => Some(r#"dotAll — "." matches newlines"#),
"u" => Some("unicode"),
"y" => Some("sticky — match at lastIndex"),
"d" => Some("indices — expose match boundaries"),
_ => None,
}
}
/// Description for a backslash escape sequence, or `None` for an unknown one.
fn escape_desc(c: char) -> Option<&'static str> {
match c {
'd' => Some("a digit [0-9]"),
'D' => Some("a non-digit"),
'w' => Some("a word character [A-Za-z0-9_]"),
'W' => Some("a non-word character"),
's' => Some("a whitespace character"),
'S' => Some("a non-whitespace character"),
'b' => Some("a word boundary"),
'B' => Some("a non-word boundary"),
'n' => Some("a newline"),
't' => Some("a tab"),
'r' => Some("a carriage return"),
_ => None,
}
}
/// Explain a regex pattern + flags into a flat list of labeled tokens.
///
/// On validation failure, returns an `ExplainResult` with `ok: false` and a
/// populated `error`; otherwise `ok: true` with the token and flag lists.
pub fn explain_regex(pattern: &str, flags: &str) -> ExplainResult {
// Collect into a Vec<char> so we can index by character position (Rust
// strings cannot be indexed by integer because they are UTF-8).
let chars: Vec<char> = pattern.chars().collect();
// Lightweight structural validation (see the module-level engine note).
if let Err(msg) = validate_structure(&chars) {
return ExplainResult {
ok: false,
tokens: Vec::new(),
flags: Vec::new(),
error: Some(msg),
};
}
let mut tokens: Vec<RegexToken> = Vec::new();
let mut push = |token: String, description: String, tokens: &mut Vec<_>| {
tokens.push(RegexToken { token, description });
};
let mut i = 0;
while i < chars.len() {
let ch = chars[i];
// --- Anchors / single-char metacharacters ------------------------
match ch {
'^' => {
push("^".into(), "start of the string (or line with /m)".into(), &mut tokens);
i += 1;
continue;
}
'$' => {
push("$".into(), "end of the string (or line with /m)".into(), &mut tokens);
i += 1;
continue;
}
'.' => {
push(".".into(), "any character (except newline, unless /s)".into(), &mut tokens);
i += 1;
continue;
}
'|' => {
push("|".into(), "OR — alternation between groups".into(), &mut tokens);
i += 1;
continue;
}
_ => {}
}
// --- Backslash escape sequences ----------------------------------
if ch == '\\' {
// A trailing lone backslash is structurally invalid and is caught
// by validate_structure; next char therefore always exists here.
let next = chars[i + 1];
let desc = match escape_desc(next) {
Some(d) => d.to_string(),
None => format!(r#"an escaped literal "{}""#, next),
};
push(format!("\\{}", next), desc, &mut tokens);
i += 2;
continue;
}
// --- Character class [ ... ] -------------------------------------
if ch == '[' {
// Unwrap is safe: validation guaranteed a matching ']'.
let end = find_class_end(&chars, i).unwrap();
let cls: String = chars[i..=end].iter().collect();
let negated = chars.get(i + 1) == Some(&'^');
let inner_start = i + 1 + if negated { 1 } else { 0 };
let inner: String = chars[inner_start..end].iter().collect();
let qualifier = if negated { "character NOT in" } else { "of" };
push(
cls,
format!("match any {}: {}", qualifier, describe_class(&inner)),
&mut tokens,
);
i = end + 1;
continue;
}
// --- Group ( ... ) — capturing, non-capturing, lookaround ---------
if ch == '(' {
let end = find_group_end(&chars, i).unwrap();
let grp: String = chars[i..=end].iter().collect();
push(grp.clone(), describe_group(&grp), &mut tokens);
i = end + 1;
continue;
}
// --- Quantifiers that attach to the previous token ----------------
if ch == '*' || ch == '+' || ch == '?' {
let lazy = chars.get(i + 1) == Some(&'?');
let base = match ch {
'*' => "0 or more times",
'+' => "1 or more times",
'?' => "0 or 1 time (optional)",
_ => unreachable!("guarded by outer if"),
};
let suffix = if lazy { " (lazy/non-greedy)" } else { " (greedy)" };
let token = if lazy { format!("{}?", ch) } else { ch.to_string() };
push(token, format!("quantifier — {}{}", base, suffix), &mut tokens);
i += if lazy { 2 } else { 1 };
continue;
}
if ch == '{' {
if let Some(end) = find_brace_end(&chars, i) {
let q: String = chars[i..=end].iter().collect();
let lazy = chars.get(end + 1) == Some(&'?');
let token = if lazy { format!("{}?", q) } else { q.clone() };
let suffix = if lazy { " (lazy)" } else { "" };
// Inner content (digits/commas) sits between the braces.
let inner: String = chars[i + 1..end].iter().collect();
push(token, format!("quantifier — repeat {} time(s){}", inner, suffix), &mut tokens);
i = end + 1 + if lazy { 1 } else { 0 };
continue;
}
// No closing brace: fall through and treat '{' as a literal.
}
// --- Default: a literal character --------------------------------
push(
ch.to_string(),
format!(r#"the literal "{}""#, escape_htmlish(ch)),
&mut tokens,
);
i += 1;
}
// Describe each flag, surfacing unknown flags explicitly.
let flag_list: Vec<FlagInfo> = flags
.chars()
.map(|f| {
let description = describe_flag(f.encode_utf8(&mut [0u8; 4]))
.map(String::from)
.unwrap_or_else(|| format!(r#"unknown flag "{}""#, f));
FlagInfo {
flag: f.to_string(),
description,
}
})
.collect();
ExplainResult {
ok: true,
tokens,
flags: flag_list,
error: None,
}
}
/// Structural validation in lieu of a regex engine (see module docs).
///
/// Checks that every `[` has a matching `]`, every `(` has a matching `)`, and
/// there is no trailing unescaped `\`. Returns `Err(message)` on the first
/// problem found.
fn validate_structure(chars: &[char]) -> Result<(), String> {
let mut i = 0;
while i < chars.len() {
match chars[i] {
'\\' => {
// Skip the escaped char; a backslash with nothing after it is
// an unterminated escape.
if i + 1 >= chars.len() {
return Err("unterminated escape: trailing backslash".to_string());
}
i += 2;
}
'[' => match find_class_end(chars, i) {
Some(end) => i = end + 1,
None => return Err("unterminated character class".to_string()),
},
'(' => match find_group_end(chars, i) {
Some(end) => i = end + 1,
None => return Err("unterminated group".to_string()),
},
_ => i += 1,
}
}
Ok(())
}
/// Index of the `]` closing the class opened at `start`, or `None` if none.
/// A leading `]` (right after `[` or `[^`) counts as a literal member.
fn find_class_end(chars: &[char], start: usize) -> Option<usize> {
let mut i = start + 1;
if i < chars.len() && chars[i] == '^' {
i += 1;
}
if i < chars.len() && chars[i] == ']' {
i += 1; // leading ] is a literal, not a terminator
}
while i < chars.len() {
if chars[i] == ']' {
return Some(i);
}
if chars[i] == '\\' {
i += 1; // skip the escaped char
}
i += 1;
}
None
}
/// Index of the `)` matching the group opened at `start`, or `None` if none.
/// Tracks nesting, skips character classes wholesale, and skips escapes.
fn find_group_end(chars: &[char], start: usize) -> Option<usize> {
let mut depth = 1usize;
let mut i = start + 1;
while i < chars.len() && depth > 0 {
match chars[i] {
'\\' => {
i += 2;
continue;
}
'[' => {
match find_class_end(chars, i) {
Some(end) => i = end + 1,
None => return None,
}
continue;
}
'(' => depth += 1,
')' => {
depth -= 1;
if depth == 0 {
return Some(i);
}
}
_ => {}
}
i += 1;
}
None
}
/// Index of the `}` closing a `{...}` quantifier opened at `start`, or `None`.
/// Unlike classes/groups, an unmatched `{` is a literal rather than an error,
/// so callers decide what to do with `None`.
fn find_brace_end(chars: &[char], start: usize) -> Option<usize> {
let mut i = start + 1;
while i < chars.len() {
if chars[i] == '}' {
return Some(i);
}
i += 1;
}
None
}
/// Render the inside of a character class for display.
fn describe_class(inner: &str) -> String {
if inner.is_empty() {
return "(empty)".to_string();
}
// Double backslashes so they display as a single literal backslash.
inner.replace('\\', "\\\\")
}
/// Classify a group by its opening syntax.
fn describe_group(grp: &str) -> String {
if grp.starts_with("(?<!") {
"lookbehind assertion (negative)".to_string()
} else if grp.starts_with("(?<=") {
"lookbehind assertion (positive)".to_string()
} else if grp.starts_with("(?:") {
"non-capturing group".to_string()
} else if grp.starts_with("(?=") {
"lookahead assertion (positive)".to_string()
} else if grp.starts_with("(?!") {
"lookahead assertion (negative)".to_string()
} else {
"capturing group".to_string()
}
}
/// Neutralize a double quote so it renders inside a quoted description.
fn escape_htmlish(c: char) -> String {
if c == '"' {
r#"\""#.to_string()
} else {
c.to_string()
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →