Skip to content

Regex Explainer — C# source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.

// regex-explainer — C# port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with
// System.Text.RegularExpressions — near-JS syntax; JS-only constructs report ok=false.
using System;
using System.Collections.Generic;
using System.Text.RegularExpressions;

var r = RegexExplainer.ExplainRegex(@"^(\w+)@([\w.-]+)$", "gi");
if (!r.Ok) { Console.WriteLine("error: " + r.Error); return 1; }
foreach (var t in r.Tokens) Console.WriteLine($"{t.Token,-14} {t.Description}");
foreach (var f in r.Flags) Console.WriteLine($"flag {f.Flag}: {f.Description}");
return 0;

namespace RegexExplainer
{
    record RegexToken(string Token, string Description);
    record FlagInfo(string Flag, string Description);
    record ExplainResult(bool Ok, List<RegexToken> Tokens, List<FlagInfo> Flags, string? Error);

    static class RegexExplainer
    {
        static readonly Dictionary<string, string> FlagDesc = new()
        {
            ["g"] = "global - find all matches", ["i"] = "case-insensitive",
            ["m"] = "multiline (^ and $ match line boundaries)", ["s"] = "dotAll - \".\" matches newlines",
            ["u"] = "unicode", ["y"] = "sticky - match at lastIndex", ["d"] = "indices - expose match boundaries",
        };

        static readonly Dictionary<string, string> EscapeDesc = new()
        {
            ["d"] = "a digit [0-9]", ["D"] = "a non-digit", ["w"] = "a word character [A-Za-z0-9_]",
            ["W"] = "a non-word character", ["s"] = "a whitespace character", ["S"] = "a non-whitespace character",
            ["b"] = "a word boundary", ["B"] = "a non-word boundary", ["n"] = "a newline",
            ["t"] = "a tab", ["r"] = "a carriage return",
        };

        /// <summary>Explain a regex pattern + flags into tokens. Never throws.</summary>
        public static ExplainResult ExplainRegex(string pattern, string flags = "")
        {
            // Validate with the native engine first (JS i/m/s map onto RegexOptions).
            var opts = RegexOptions.None;
            foreach (var f in flags)
            {
                if (f == 'i') opts |= RegexOptions.IgnoreCase;
                if (f == 'm') opts |= RegexOptions.Multiline;
                if (f == 's') opts |= RegexOptions.Singleline;
            }
            try { _ = new Regex(pattern, opts); }
            catch (ArgumentException e) { return new ExplainResult(false, new(), new(), e.Message); }

            var tokens = new List<RegexToken>();
            void Push(string token, string description) => tokens.Add(new(token, description));
            var p = pattern;
            var i = 0;
            while (i < p.Length)
            {
                var ch = p[i];
                switch (ch)
                {
                    case '^': Push("^", "start of the string (or line with /m)"); i++; break;
                    case '$': Push("$", "end of the string (or line with /m)"); i++; break;
                    case '.': Push(".", "any character (except newline, unless /s)"); i++; break;
                    case '|': Push("|", "OR - alternation between groups"); i++; break;
                    case '\\':
                    {
                        var next = i + 1 < p.Length ? p[i + 1].ToString() : "";
                        Push("\\" + next, EscapeDesc.GetValueOrDefault(next, $"an escaped literal \"{next}\""));
                        i += 2;
                        break;
                    }
                    case '[':
                    {
                        var end = FindClassEnd(p, i);
                        var cls = p[i..(end + 1)];
                        var negated = p[i + 1] == '^';
                        var inner = cls[(1 + (negated ? 1 : 0))..^1];
                        Push(cls, $"match any {(negated ? "character NOT in" : "of")}: {DescribeClass(inner)}");
                        i = end + 1;
                        break;
                    }
                    case '(':
                    {
                        var end = FindGroupEnd(p, i);
                        var grp = p[i..(end + 1)];
                        Push(grp, DescribeGroup(grp));
                        i = end + 1;
                        break;
                    }
                    case '*' or '+' or '?':
                    {
                        var lazy = p[i + 1] == '?';
                        var @base = ch switch { '*' => "0 or more times", '+' => "1 or more times", _ => "0 or 1 time (optional)" };
                        Push(ch.ToString() + (lazy ? "?" : ""),
                             $"quantifier - {@base}{(lazy ? " (lazy/non-greedy)" : " (greedy)")}");
                        i += lazy ? 2 : 1;
                        break;
                    }
                    case '{' when p.IndexOf('}', i) != -1:
                    {
                        var end = p.IndexOf('}', i);
                        var lazy = end + 1 < p.Length && p[end + 1] == '?';
                        var q = p[i..(end + 1)];
                        Push(q + (lazy ? "?" : ""),
                             $"quantifier - repeat {q[1..^1]} time(s){(lazy ? " (lazy)" : "")}");
                        i = end + 1 + (lazy ? 1 : 0);
                        break;
                    }
                    default: // a literal character (also '{' with no closing brace)
                        Push(ch.ToString(), $"the literal \"{ch}\"");
                        i++;
                        break;
                }
            }

            var flagList = new List<FlagInfo>();
            foreach (var f in flags)
                flagList.Add(new FlagInfo(f.ToString(),
                    FlagDesc.GetValueOrDefault(f.ToString(), $"unknown flag \"{f}\"")));
            return new ExplainResult(true, tokens, flagList, null);
        }

        /// <summary>Index of the ']' closing a class opened at start; a leading ']' is literal.</summary>
        static int FindClassEnd(string p, int start)
        {
            var i = start + 1;
            if (i < p.Length && p[i] == '^') i++;
            if (i < p.Length && p[i] == ']') i++;
            while (i < p.Length && p[i] != ']') { if (p[i] == '\\') i++; i++; }
            return i < p.Length ? i : p.Length - 1;
        }

        /// <summary>Index of the ')' matching the group opened at start; skips classes + escapes.</summary>
        static int FindGroupEnd(string p, int start)
        {
            var depth = 1; var i = start + 1;
            while (i < p.Length && depth > 0)
            {
                if (p[i] == '\\') { i += 2; continue; }
                if (p[i] == '[') { i = FindClassEnd(p, i) + 1; continue; }
                if (p[i] == '(') depth++;
                else if (p[i] == ')') depth--;
                i++;
            }
            return i - 1;
        }

        static string DescribeClass(string inner) =>
            inner.Length == 0 ? "(empty)" : inner.Replace("\\", "\\\\");

        static string DescribeGroup(string grp) => grp switch
        {
            _ when grp.StartsWith("(?:") => "non-capturing group",
            _ when grp.StartsWith("(?=") => "lookahead assertion (positive)",
            _ when grp.StartsWith("(?!") => "lookahead assertion (negative)",
            _ when grp.StartsWith("(?<=") => "lookbehind assertion (positive)",
            _ when grp.StartsWith("(?<!") => "lookbehind assertion (negative)",
            _ => "capturing group",
        };
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →