Skip to content

Regex Explainer — Rust source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! Regex Explainer — Rust port.
//!
//! Language:    Rust (std-only — no external crates).
//! Source:      CosmoDev polyglot showcase port of the `regex-explainer` tool.
//! Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
//!
//! What it does:
//!   Tokenizes a regular expression into labeled tokens — anchors, escapes,
//!   character classes, quantifiers, groups, alternation, and literals — and
//!   describes each flag. The explainer never panics: an invalid pattern yields
//!   a structured error result.
//!
//! Engine note: Rust's standard library has no regular-expression engine, and
//! pulling in the external `regex` crate would violate the showcase's
//! dependency-free contract. The TS original validates by asking the JS regex
//! engine to compile the pattern; here we instead perform a lightweight
//! *structural* validation (balanced `[...]` and `(...)`, no trailing `\`).
//! This catches the common malformed-input cases without false negatives on
//! the syntactic constructs this tokenizer recognizes (lookaround,
//! backreferences, etc., which a real engine might still reject).
//!
//! This is display source — part of CosmoDev's polyglot tool pages.

/// A labeled slice of a regex pattern.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct RegexToken {
    pub token: String,
    pub description: String,
}

/// A flag letter paired with its human description.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct FlagInfo {
    pub flag: String,
    pub description: String,
}

/// The full output of [`explain_regex`].
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct ExplainResult {
    pub ok: bool,
    pub tokens: Vec<RegexToken>,
    pub flags: Vec<FlagInfo>,
    pub error: Option<String>,
}

/// Human description for a single flag letter, or `None` if unrecognized.
/// Mirrors the TS `describeFlag` (returns `string | null`).
pub fn describe_flag(flag: &str) -> Option<&'static str> {
    match flag {
        "g" => Some("global — find all matches"),
        "i" => Some("case-insensitive"),
        "m" => Some("multiline (^ and $ match line boundaries)"),
        "s" => Some(r#"dotAll — "." matches newlines"#),
        "u" => Some("unicode"),
        "y" => Some("sticky — match at lastIndex"),
        "d" => Some("indices — expose match boundaries"),
        _ => None,
    }
}

/// Description for a backslash escape sequence, or `None` for an unknown one.
fn escape_desc(c: char) -> Option<&'static str> {
    match c {
        'd' => Some("a digit [0-9]"),
        'D' => Some("a non-digit"),
        'w' => Some("a word character [A-Za-z0-9_]"),
        'W' => Some("a non-word character"),
        's' => Some("a whitespace character"),
        'S' => Some("a non-whitespace character"),
        'b' => Some("a word boundary"),
        'B' => Some("a non-word boundary"),
        'n' => Some("a newline"),
        't' => Some("a tab"),
        'r' => Some("a carriage return"),
        _ => None,
    }
}

/// Explain a regex pattern + flags into a flat list of labeled tokens.
///
/// On validation failure, returns an `ExplainResult` with `ok: false` and a
/// populated `error`; otherwise `ok: true` with the token and flag lists.
pub fn explain_regex(pattern: &str, flags: &str) -> ExplainResult {
    // Collect into a Vec<char> so we can index by character position (Rust
    // strings cannot be indexed by integer because they are UTF-8).
    let chars: Vec<char> = pattern.chars().collect();

    // Lightweight structural validation (see the module-level engine note).
    if let Err(msg) = validate_structure(&chars) {
        return ExplainResult {
            ok: false,
            tokens: Vec::new(),
            flags: Vec::new(),
            error: Some(msg),
        };
    }

    let mut tokens: Vec<RegexToken> = Vec::new();
    let mut push = |token: String, description: String, tokens: &mut Vec<_>| {
        tokens.push(RegexToken { token, description });
    };

    let mut i = 0;
    while i < chars.len() {
        let ch = chars[i];

        // --- Anchors / single-char metacharacters ------------------------
        match ch {
            '^' => {
                push("^".into(), "start of the string (or line with /m)".into(), &mut tokens);
                i += 1;
                continue;
            }
            '$' => {
                push("$".into(), "end of the string (or line with /m)".into(), &mut tokens);
                i += 1;
                continue;
            }
            '.' => {
                push(".".into(), "any character (except newline, unless /s)".into(), &mut tokens);
                i += 1;
                continue;
            }
            '|' => {
                push("|".into(), "OR — alternation between groups".into(), &mut tokens);
                i += 1;
                continue;
            }
            _ => {}
        }

        // --- Backslash escape sequences ----------------------------------
        if ch == '\\' {
            // A trailing lone backslash is structurally invalid and is caught
            // by validate_structure; next char therefore always exists here.
            let next = chars[i + 1];
            let desc = match escape_desc(next) {
                Some(d) => d.to_string(),
                None => format!(r#"an escaped literal "{}""#, next),
            };
            push(format!("\\{}", next), desc, &mut tokens);
            i += 2;
            continue;
        }

        // --- Character class [ ... ] -------------------------------------
        if ch == '[' {
            // Unwrap is safe: validation guaranteed a matching ']'.
            let end = find_class_end(&chars, i).unwrap();
            let cls: String = chars[i..=end].iter().collect();
            let negated = chars.get(i + 1) == Some(&'^');
            let inner_start = i + 1 + if negated { 1 } else { 0 };
            let inner: String = chars[inner_start..end].iter().collect();
            let qualifier = if negated { "character NOT in" } else { "of" };
            push(
                cls,
                format!("match any {}: {}", qualifier, describe_class(&inner)),
                &mut tokens,
            );
            i = end + 1;
            continue;
        }

        // --- Group ( ... ) — capturing, non-capturing, lookaround ---------
        if ch == '(' {
            let end = find_group_end(&chars, i).unwrap();
            let grp: String = chars[i..=end].iter().collect();
            push(grp.clone(), describe_group(&grp), &mut tokens);
            i = end + 1;
            continue;
        }

        // --- Quantifiers that attach to the previous token ----------------
        if ch == '*' || ch == '+' || ch == '?' {
            let lazy = chars.get(i + 1) == Some(&'?');
            let base = match ch {
                '*' => "0 or more times",
                '+' => "1 or more times",
                '?' => "0 or 1 time (optional)",
                _ => unreachable!("guarded by outer if"),
            };
            let suffix = if lazy { " (lazy/non-greedy)" } else { " (greedy)" };
            let token = if lazy { format!("{}?", ch) } else { ch.to_string() };
            push(token, format!("quantifier — {}{}", base, suffix), &mut tokens);
            i += if lazy { 2 } else { 1 };
            continue;
        }
        if ch == '{' {
            if let Some(end) = find_brace_end(&chars, i) {
                let q: String = chars[i..=end].iter().collect();
                let lazy = chars.get(end + 1) == Some(&'?');
                let token = if lazy { format!("{}?", q) } else { q.clone() };
                let suffix = if lazy { " (lazy)" } else { "" };
                // Inner content (digits/commas) sits between the braces.
                let inner: String = chars[i + 1..end].iter().collect();
                push(token, format!("quantifier — repeat {} time(s){}", inner, suffix), &mut tokens);
                i = end + 1 + if lazy { 1 } else { 0 };
                continue;
            }
            // No closing brace: fall through and treat '{' as a literal.
        }

        // --- Default: a literal character --------------------------------
        push(
            ch.to_string(),
            format!(r#"the literal "{}""#, escape_htmlish(ch)),
            &mut tokens,
        );
        i += 1;
    }

    // Describe each flag, surfacing unknown flags explicitly.
    let flag_list: Vec<FlagInfo> = flags
        .chars()
        .map(|f| {
            let description = describe_flag(f.encode_utf8(&mut [0u8; 4]))
                .map(String::from)
                .unwrap_or_else(|| format!(r#"unknown flag "{}""#, f));
            FlagInfo {
                flag: f.to_string(),
                description,
            }
        })
        .collect();

    ExplainResult {
        ok: true,
        tokens,
        flags: flag_list,
        error: None,
    }
}

/// Structural validation in lieu of a regex engine (see module docs).
///
/// Checks that every `[` has a matching `]`, every `(` has a matching `)`, and
/// there is no trailing unescaped `\`. Returns `Err(message)` on the first
/// problem found.
fn validate_structure(chars: &[char]) -> Result<(), String> {
    let mut i = 0;
    while i < chars.len() {
        match chars[i] {
            '\\' => {
                // Skip the escaped char; a backslash with nothing after it is
                // an unterminated escape.
                if i + 1 >= chars.len() {
                    return Err("unterminated escape: trailing backslash".to_string());
                }
                i += 2;
            }
            '[' => match find_class_end(chars, i) {
                Some(end) => i = end + 1,
                None => return Err("unterminated character class".to_string()),
            },
            '(' => match find_group_end(chars, i) {
                Some(end) => i = end + 1,
                None => return Err("unterminated group".to_string()),
            },
            _ => i += 1,
        }
    }
    Ok(())
}

/// Index of the `]` closing the class opened at `start`, or `None` if none.
/// A leading `]` (right after `[` or `[^`) counts as a literal member.
fn find_class_end(chars: &[char], start: usize) -> Option<usize> {
    let mut i = start + 1;
    if i < chars.len() && chars[i] == '^' {
        i += 1;
    }
    if i < chars.len() && chars[i] == ']' {
        i += 1; // leading ] is a literal, not a terminator
    }
    while i < chars.len() {
        if chars[i] == ']' {
            return Some(i);
        }
        if chars[i] == '\\' {
            i += 1; // skip the escaped char
        }
        i += 1;
    }
    None
}

/// Index of the `)` matching the group opened at `start`, or `None` if none.
/// Tracks nesting, skips character classes wholesale, and skips escapes.
fn find_group_end(chars: &[char], start: usize) -> Option<usize> {
    let mut depth = 1usize;
    let mut i = start + 1;
    while i < chars.len() && depth > 0 {
        match chars[i] {
            '\\' => {
                i += 2;
                continue;
            }
            '[' => {
                match find_class_end(chars, i) {
                    Some(end) => i = end + 1,
                    None => return None,
                }
                continue;
            }
            '(' => depth += 1,
            ')' => {
                depth -= 1;
                if depth == 0 {
                    return Some(i);
                }
            }
            _ => {}
        }
        i += 1;
    }
    None
}

/// Index of the `}` closing a `{...}` quantifier opened at `start`, or `None`.
/// Unlike classes/groups, an unmatched `{` is a literal rather than an error,
/// so callers decide what to do with `None`.
fn find_brace_end(chars: &[char], start: usize) -> Option<usize> {
    let mut i = start + 1;
    while i < chars.len() {
        if chars[i] == '}' {
            return Some(i);
        }
        i += 1;
    }
    None
}

/// Render the inside of a character class for display.
fn describe_class(inner: &str) -> String {
    if inner.is_empty() {
        return "(empty)".to_string();
    }
    // Double backslashes so they display as a single literal backslash.
    inner.replace('\\', "\\\\")
}

/// Classify a group by its opening syntax.
fn describe_group(grp: &str) -> String {
    if grp.starts_with("(?<!") {
        "lookbehind assertion (negative)".to_string()
    } else if grp.starts_with("(?<=") {
        "lookbehind assertion (positive)".to_string()
    } else if grp.starts_with("(?:") {
        "non-capturing group".to_string()
    } else if grp.starts_with("(?=") {
        "lookahead assertion (positive)".to_string()
    } else if grp.starts_with("(?!") {
        "lookahead assertion (negative)".to_string()
    } else {
        "capturing group".to_string()
    }
}

/// Neutralize a double quote so it renders inside a quoted description.
fn escape_htmlish(c: char) -> String {
    if c == '"' {
        r#"\""#.to_string()
    } else {
        c.to_string()
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →