Regex Explainer — Go source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.
// Package regexexplainer is the Go twin of CosmoDev's src/lib/regexExplain.ts
// (dual source: the web lib is TypeScript, the CLI lib is Go — kept in
// lock-step). Pure + deterministic, never panics. The table-driven tests in
// regex-explainer_test.go share vectors with src/lib/regexExplain.test.ts so
// the two implementations are held to the same contract.
//
// The twin tokenizes a regular expression into labeled tokens (anchors,
// escapes, classes, quantifiers, groups, alternation, literals) and describes
// the flags. The algorithm mirrors the TS lib's hand-written scanner exactly:
// validate → walk the pattern rune by rune, slicing out whole constructs via
// findClassEnd/findGroupEnd → emit one RegexToken per construct → describe each
// flag. It does NOT execute the pattern.
//
// Note on validation: the TS lib delegates validity checking to the JavaScript
// RegExp constructor (new RegExp(pattern, flags)). Go's regexp package cannot
// stand in for that because RE2 rejects constructs the tokenizer must accept
// (lookahead/lookbehind, backreferences). validateRegex therefore performs a
// structural scan for the syntactic errors JS raises on the shared vectors and
// common inputs: invalid flags, a trailing backslash, an unterminated character
// class, and an unterminated group. This agrees with JS on every shared test
// vector; it is not a complete JS regex grammar.
package regexexplainer
import (
"fmt"
"strings"
)
// RegexToken is a single labeled slice of the pattern.
type RegexToken struct {
Token string
Description string
}
// FlagDesc is one flag character paired with its description.
type FlagDesc struct {
Flag string
Description string
}
// ExplainResult is the outcome of explaining a pattern. Error is "" when there
// is none (mirroring the TS lib's null), and OK is false with an empty token /
// flag list when the pattern is invalid.
type ExplainResult struct {
OK bool
Tokens []RegexToken
Flags []FlagDesc
Error string
}
// flagDesc maps a flag character to its human description, mirroring FLAG_DESC.
var flagDesc = map[rune]string{
'g': "global — find all matches",
'i': "case-insensitive",
'm': "multiline — ^ and $ match line boundaries",
's': "dotAll — \".\" matches newlines",
'u': "unicode",
'y': "sticky — match at lastIndex",
'd': "indices — expose match boundaries",
}
// DescribeFlag returns the description for a single flag character and ok=true,
// or ok=false when the flag is unknown. It is the Go twin of describeFlag()
// (which returns string | null).
func DescribeFlag(flag rune) (string, bool) {
d, ok := flagDesc[flag]
return d, ok
}
// escapeDesc maps an escaped character to its description, mirroring ESCAPE_DESC.
var escapeDesc = map[string]string{
"d": "a digit [0-9]",
"D": "a non-digit",
"w": "a word character [A-Za-z0-9_]",
"W": "a non-word character",
"s": "a whitespace character",
"S": "a non-whitespace character",
"b": "a word boundary",
"B": "a non-word boundary",
"n": "a newline",
"t": "a tab",
"r": "a carriage return",
}
// escapeHtmlish mirrors the TS helper: it escapes a literal double-quote so the
// rendered description reads as the literal "\"". Used only for literal tokens.
func escapeHtmlish(s string) string {
return strings.ReplaceAll(s, "\"", "\\\"")
}
// validFlags is the set of flag characters the JavaScript RegExp constructor
// accepts (ES2024 Unicode Sets adds 'v'). Used by validateRegex.
const validFlags = "dgimsuvy"
// validateRegex mirrors the validity check the TS lib delegates to the
// JavaScript RegExp constructor. It returns a non-empty error string when the
// pattern or flags would be rejected, and "" when valid. See the package doc
// comment for why Go's regexp package cannot be used here.
func validateRegex(pattern, flags string) string {
for _, f := range flags {
if !strings.ContainsRune(validFlags, f) {
return fmt.Sprintf("Invalid flags %q", flags)
}
}
runes := []rune(pattern)
n := len(runes)
i := 0
groupDepth := 0
for i < n {
switch runes[i] {
case '\\':
// A backslash must be followed by the character it escapes.
if i+1 >= n {
return "\\ at end of pattern"
}
i += 2
case '[':
// For VALIDITY (distinct from findClassEnd's tokenization), a class
// is closed by the next unescaped ']'. A leading ']' is a literal
// member under JS annex-B, but that only affects where the class
// ENDS for display — an unescaped ']' anywhere still closes it. So
// "[]" and "[^]" are valid empty classes, while "[" and "[a\]" are
// unterminated. Backslash escapes the next character.
i++
closed := false
for i < n {
if runes[i] == '\\' {
i += 2
continue
}
if runes[i] == ']' {
closed = true
i++
break
}
i++
}
if !closed {
return "Unterminated character class"
}
case '(':
groupDepth++
i++
case ')':
if groupDepth == 0 {
return "Unmatched ')'"
}
groupDepth--
i++
default:
i++
}
}
if groupDepth > 0 {
return "Unterminated group"
}
return ""
}
// findClassEnd returns the index of the ']' that closes the character class
// opening at start. It mirrors findClassEnd() in the TS lib, including the
// literal-member rules for a leading '^' and ']'. If no closing ']' exists it
// returns the last index of the slice (mirroring the TS `p.length - 1`).
func findClassEnd(runes []rune, start int) int {
n := len(runes)
i := start + 1
if i < n && runes[i] == '^' {
i++
}
if i < n && runes[i] == ']' {
i++
}
for i < n && runes[i] != ']' {
if runes[i] == '\\' {
i++
}
i++
}
if i < n {
return i
}
return n - 1
}
// findGroupEnd returns the index of the ')' that closes the group opening at
// start, honoring nested groups, escapes, and character classes. It mirrors
// findGroupEnd() in the TS lib. If the group never closes it returns the last
// index walked (mirroring the TS `i - 1`).
func findGroupEnd(runes []rune, start int) int {
n := len(runes)
depth := 1
i := start + 1
for i < n && depth > 0 {
switch runes[i] {
case '\\':
i += 2
continue
case '[':
i = findClassEnd(runes, i) + 1
continue
case '(':
depth++
case ')':
depth--
}
i++
}
return i - 1
}
// describeClass renders the inner text of a character class. It mirrors
// describeClass() in the TS lib: an empty inner set is "(empty)", otherwise
// each backslash is doubled for display.
func describeClass(inner string) string {
if inner == "" {
return "(empty)"
}
return strings.ReplaceAll(inner, "\\", "\\\\")
}
// describeGroup classifies a group token by its opening syntax. It mirrors
// describeGroup() in the TS lib.
func describeGroup(grp string) string {
switch {
case strings.HasPrefix(grp, "(?:"):
return "non-capturing group"
case strings.HasPrefix(grp, "(?="):
return "lookahead assertion (positive)"
case strings.HasPrefix(grp, "(?!"):
return "lookahead assertion (negative)"
case strings.HasPrefix(grp, "(?<="):
return "lookbehind assertion (positive)"
case strings.HasPrefix(grp, "(?<!"):
return "lookbehind assertion (negative)"
default:
return "capturing group"
}
}
// ExplainRegex explains a regex pattern and its flags into labeled tokens. It
// is the Go twin of explainRegex() in src/lib/regexExplain.ts and must agree
// with it on every shared vector. Never panics: an invalid pattern yields
// ExplainResult{OK: false, Error: ...} with empty tokens and flags.
func ExplainRegex(pattern, flags string) ExplainResult {
if msg := validateRegex(pattern, flags); msg != "" {
return ExplainResult{OK: false, Error: msg}
}
runes := []rune(pattern)
n := len(runes)
tokens := make([]RegexToken, 0)
i := 0
for i < n {
ch := runes[i]
switch ch {
case '^':
tokens = append(tokens, RegexToken{"^", "start of the string (or line with /m)"})
i++
case '$':
tokens = append(tokens, RegexToken{"$", "end of the string (or line with /m)"})
i++
case '.':
tokens = append(tokens, RegexToken{".", "any character (except newline, unless /s)"})
i++
case '|':
tokens = append(tokens, RegexToken{"|", "OR — alternation between groups"})
i++
case '\\':
next := ""
if i+1 < n {
next = string(runes[i+1])
}
desc, ok := escapeDesc[next]
if !ok {
desc = fmt.Sprintf("an escaped literal \"%s\"", next)
}
tokens = append(tokens, RegexToken{"\\" + next, desc})
i += 2
case '[':
end := findClassEnd(runes, i)
cls := string(runes[i : end+1])
negated := i+1 < n && runes[i+1] == '^'
innerStart := i + 1
if negated {
innerStart++
}
inner := ""
if innerStart < end {
inner = string(runes[innerStart:end])
}
kind := "of"
if negated {
kind = "character NOT in"
}
tokens = append(tokens, RegexToken{cls, fmt.Sprintf("match any %s: %s", kind, describeClass(inner))})
i = end + 1
case '(':
end := findGroupEnd(runes, i)
grp := string(runes[i : end+1])
tokens = append(tokens, RegexToken{grp, describeGroup(grp)})
i = end + 1
case '*', '+', '?':
lazy := i+1 < n && runes[i+1] == '?'
base := ""
switch ch {
case '*':
base = "0 or more times"
case '+':
base = "1 or more times"
case '?':
base = "0 or 1 time (optional)"
}
suffix := " (greedy)"
if lazy {
suffix = " (lazy/non-greedy)"
}
tok := string(ch)
if lazy {
tok += "?"
}
tokens = append(tokens, RegexToken{tok, "quantifier — " + base + suffix})
if lazy {
i += 2
} else {
i++
}
case '{':
// Find the closing '}'. When there is one this is a repeat
// quantifier; otherwise the '{' is a literal (JS only treats a
// well-formed {n}, {n,}, {n,m} as a quantifier).
end := -1
for k := i; k < n; k++ {
if runes[k] == '}' {
end = k
break
}
}
if end != -1 {
q := string(runes[i : end+1])
lazy := end+1 < n && runes[end+1] == '?'
repeat := string(runes[i+1 : end]) // q without the surrounding { }
suffix := ""
if lazy {
suffix = " (lazy)"
}
tok := q
if lazy {
tok += "?"
}
tokens = append(tokens, RegexToken{tok, "quantifier — repeat " + repeat + " time(s)" + suffix})
i = end + 1
if lazy {
i++
}
} else {
tokens = append(tokens, RegexToken{string(ch), "the literal \"" + escapeHtmlish(string(ch)) + "\""})
i++
}
default:
tokens = append(tokens, RegexToken{string(ch), "the literal \"" + escapeHtmlish(string(ch)) + "\""})
i++
}
}
flagList := make([]FlagDesc, 0, len(flags))
for _, f := range flags {
d, ok := DescribeFlag(f)
if !ok {
d = fmt.Sprintf("unknown flag \"%s\"", string(f))
}
flagList = append(flagList, FlagDesc{Flag: string(f), Description: d})
}
return ExplainResult{OK: true, Tokens: tokens, Flags: flagList}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →