Regex Explainer — C# source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the C# implementation — the same logic the interactive tool runs, in a shareable, citable form.
// regex-explainer — C# port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with
// System.Text.RegularExpressions — near-JS syntax; JS-only constructs report ok=false.
using System;
using System.Collections.Generic;
using System.Text.RegularExpressions;
var r = RegexExplainer.ExplainRegex(@"^(\w+)@([\w.-]+)$", "gi");
if (!r.Ok) { Console.WriteLine("error: " + r.Error); return 1; }
foreach (var t in r.Tokens) Console.WriteLine($"{t.Token,-14} {t.Description}");
foreach (var f in r.Flags) Console.WriteLine($"flag {f.Flag}: {f.Description}");
return 0;
namespace RegexExplainer
{
record RegexToken(string Token, string Description);
record FlagInfo(string Flag, string Description);
record ExplainResult(bool Ok, List<RegexToken> Tokens, List<FlagInfo> Flags, string? Error);
static class RegexExplainer
{
static readonly Dictionary<string, string> FlagDesc = new()
{
["g"] = "global - find all matches", ["i"] = "case-insensitive",
["m"] = "multiline (^ and $ match line boundaries)", ["s"] = "dotAll - \".\" matches newlines",
["u"] = "unicode", ["y"] = "sticky - match at lastIndex", ["d"] = "indices - expose match boundaries",
};
static readonly Dictionary<string, string> EscapeDesc = new()
{
["d"] = "a digit [0-9]", ["D"] = "a non-digit", ["w"] = "a word character [A-Za-z0-9_]",
["W"] = "a non-word character", ["s"] = "a whitespace character", ["S"] = "a non-whitespace character",
["b"] = "a word boundary", ["B"] = "a non-word boundary", ["n"] = "a newline",
["t"] = "a tab", ["r"] = "a carriage return",
};
/// <summary>Explain a regex pattern + flags into tokens. Never throws.</summary>
public static ExplainResult ExplainRegex(string pattern, string flags = "")
{
// Validate with the native engine first (JS i/m/s map onto RegexOptions).
var opts = RegexOptions.None;
foreach (var f in flags)
{
if (f == 'i') opts |= RegexOptions.IgnoreCase;
if (f == 'm') opts |= RegexOptions.Multiline;
if (f == 's') opts |= RegexOptions.Singleline;
}
try { _ = new Regex(pattern, opts); }
catch (ArgumentException e) { return new ExplainResult(false, new(), new(), e.Message); }
var tokens = new List<RegexToken>();
void Push(string token, string description) => tokens.Add(new(token, description));
var p = pattern;
var i = 0;
while (i < p.Length)
{
var ch = p[i];
switch (ch)
{
case '^': Push("^", "start of the string (or line with /m)"); i++; break;
case '$': Push("$", "end of the string (or line with /m)"); i++; break;
case '.': Push(".", "any character (except newline, unless /s)"); i++; break;
case '|': Push("|", "OR - alternation between groups"); i++; break;
case '\\':
{
var next = i + 1 < p.Length ? p[i + 1].ToString() : "";
Push("\\" + next, EscapeDesc.GetValueOrDefault(next, $"an escaped literal \"{next}\""));
i += 2;
break;
}
case '[':
{
var end = FindClassEnd(p, i);
var cls = p[i..(end + 1)];
var negated = p[i + 1] == '^';
var inner = cls[(1 + (negated ? 1 : 0))..^1];
Push(cls, $"match any {(negated ? "character NOT in" : "of")}: {DescribeClass(inner)}");
i = end + 1;
break;
}
case '(':
{
var end = FindGroupEnd(p, i);
var grp = p[i..(end + 1)];
Push(grp, DescribeGroup(grp));
i = end + 1;
break;
}
case '*' or '+' or '?':
{
var lazy = p[i + 1] == '?';
var @base = ch switch { '*' => "0 or more times", '+' => "1 or more times", _ => "0 or 1 time (optional)" };
Push(ch.ToString() + (lazy ? "?" : ""),
$"quantifier - {@base}{(lazy ? " (lazy/non-greedy)" : " (greedy)")}");
i += lazy ? 2 : 1;
break;
}
case '{' when p.IndexOf('}', i) != -1:
{
var end = p.IndexOf('}', i);
var lazy = end + 1 < p.Length && p[end + 1] == '?';
var q = p[i..(end + 1)];
Push(q + (lazy ? "?" : ""),
$"quantifier - repeat {q[1..^1]} time(s){(lazy ? " (lazy)" : "")}");
i = end + 1 + (lazy ? 1 : 0);
break;
}
default: // a literal character (also '{' with no closing brace)
Push(ch.ToString(), $"the literal \"{ch}\"");
i++;
break;
}
}
var flagList = new List<FlagInfo>();
foreach (var f in flags)
flagList.Add(new FlagInfo(f.ToString(),
FlagDesc.GetValueOrDefault(f.ToString(), $"unknown flag \"{f}\"")));
return new ExplainResult(true, tokens, flagList, null);
}
/// <summary>Index of the ']' closing a class opened at start; a leading ']' is literal.</summary>
static int FindClassEnd(string p, int start)
{
var i = start + 1;
if (i < p.Length && p[i] == '^') i++;
if (i < p.Length && p[i] == ']') i++;
while (i < p.Length && p[i] != ']') { if (p[i] == '\\') i++; i++; }
return i < p.Length ? i : p.Length - 1;
}
/// <summary>Index of the ')' matching the group opened at start; skips classes + escapes.</summary>
static int FindGroupEnd(string p, int start)
{
var depth = 1; var i = start + 1;
while (i < p.Length && depth > 0)
{
if (p[i] == '\\') { i += 2; continue; }
if (p[i] == '[') { i = FindClassEnd(p, i) + 1; continue; }
if (p[i] == '(') depth++;
else if (p[i] == ')') depth--;
i++;
}
return i - 1;
}
static string DescribeClass(string inner) =>
inner.Length == 0 ? "(empty)" : inner.Replace("\\", "\\\\");
static string DescribeGroup(string grp) => grp switch
{
_ when grp.StartsWith("(?:") => "non-capturing group",
_ when grp.StartsWith("(?=") => "lookahead assertion (positive)",
_ when grp.StartsWith("(?!") => "lookahead assertion (negative)",
_ when grp.StartsWith("(?<=") => "lookbehind assertion (positive)",
_ when grp.StartsWith("(?<!") => "lookbehind assertion (negative)",
_ => "capturing group",
};
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →