Skip to content

Regex Explainer — Go source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Go implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Package regexexplainer is the Go twin of CosmoDev's src/lib/regexExplain.ts
// (dual source: the web lib is TypeScript, the CLI lib is Go — kept in
// lock-step). Pure + deterministic, never panics. The table-driven tests in
// regex-explainer_test.go share vectors with src/lib/regexExplain.test.ts so
// the two implementations are held to the same contract.
//
// The twin tokenizes a regular expression into labeled tokens (anchors,
// escapes, classes, quantifiers, groups, alternation, literals) and describes
// the flags. The algorithm mirrors the TS lib's hand-written scanner exactly:
// validate → walk the pattern rune by rune, slicing out whole constructs via
// findClassEnd/findGroupEnd → emit one RegexToken per construct → describe each
// flag. It does NOT execute the pattern.
//
// Note on validation: the TS lib delegates validity checking to the JavaScript
// RegExp constructor (new RegExp(pattern, flags)). Go's regexp package cannot
// stand in for that because RE2 rejects constructs the tokenizer must accept
// (lookahead/lookbehind, backreferences). validateRegex therefore performs a
// structural scan for the syntactic errors JS raises on the shared vectors and
// common inputs: invalid flags, a trailing backslash, an unterminated character
// class, and an unterminated group. This agrees with JS on every shared test
// vector; it is not a complete JS regex grammar.
package regexexplainer

import (
	"fmt"
	"strings"
)

// RegexToken is a single labeled slice of the pattern.
type RegexToken struct {
	Token       string
	Description string
}

// FlagDesc is one flag character paired with its description.
type FlagDesc struct {
	Flag        string
	Description string
}

// ExplainResult is the outcome of explaining a pattern. Error is "" when there
// is none (mirroring the TS lib's null), and OK is false with an empty token /
// flag list when the pattern is invalid.
type ExplainResult struct {
	OK     bool
	Tokens []RegexToken
	Flags  []FlagDesc
	Error  string
}

// flagDesc maps a flag character to its human description, mirroring FLAG_DESC.
var flagDesc = map[rune]string{
	'g': "global — find all matches",
	'i': "case-insensitive",
	'm': "multiline — ^ and $ match line boundaries",
	's': "dotAll — \".\" matches newlines",
	'u': "unicode",
	'y': "sticky — match at lastIndex",
	'd': "indices — expose match boundaries",
}

// DescribeFlag returns the description for a single flag character and ok=true,
// or ok=false when the flag is unknown. It is the Go twin of describeFlag()
// (which returns string | null).
func DescribeFlag(flag rune) (string, bool) {
	d, ok := flagDesc[flag]
	return d, ok
}

// escapeDesc maps an escaped character to its description, mirroring ESCAPE_DESC.
var escapeDesc = map[string]string{
	"d": "a digit [0-9]",
	"D": "a non-digit",
	"w": "a word character [A-Za-z0-9_]",
	"W": "a non-word character",
	"s": "a whitespace character",
	"S": "a non-whitespace character",
	"b": "a word boundary",
	"B": "a non-word boundary",
	"n": "a newline",
	"t": "a tab",
	"r": "a carriage return",
}

// escapeHtmlish mirrors the TS helper: it escapes a literal double-quote so the
// rendered description reads as the literal "\"". Used only for literal tokens.
func escapeHtmlish(s string) string {
	return strings.ReplaceAll(s, "\"", "\\\"")
}

// validFlags is the set of flag characters the JavaScript RegExp constructor
// accepts (ES2024 Unicode Sets adds 'v'). Used by validateRegex.
const validFlags = "dgimsuvy"

// validateRegex mirrors the validity check the TS lib delegates to the
// JavaScript RegExp constructor. It returns a non-empty error string when the
// pattern or flags would be rejected, and "" when valid. See the package doc
// comment for why Go's regexp package cannot be used here.
func validateRegex(pattern, flags string) string {
	for _, f := range flags {
		if !strings.ContainsRune(validFlags, f) {
			return fmt.Sprintf("Invalid flags %q", flags)
		}
	}

	runes := []rune(pattern)
	n := len(runes)
	i := 0
	groupDepth := 0
	for i < n {
		switch runes[i] {
		case '\\':
			// A backslash must be followed by the character it escapes.
			if i+1 >= n {
				return "\\ at end of pattern"
			}
			i += 2
		case '[':
			// For VALIDITY (distinct from findClassEnd's tokenization), a class
			// is closed by the next unescaped ']'. A leading ']' is a literal
			// member under JS annex-B, but that only affects where the class
			// ENDS for display — an unescaped ']' anywhere still closes it. So
			// "[]" and "[^]" are valid empty classes, while "[" and "[a\]" are
			// unterminated. Backslash escapes the next character.
			i++
			closed := false
			for i < n {
				if runes[i] == '\\' {
					i += 2
					continue
				}
				if runes[i] == ']' {
					closed = true
					i++
					break
				}
				i++
			}
			if !closed {
				return "Unterminated character class"
			}
		case '(':
			groupDepth++
			i++
		case ')':
			if groupDepth == 0 {
				return "Unmatched ')'"
			}
			groupDepth--
			i++
		default:
			i++
		}
	}
	if groupDepth > 0 {
		return "Unterminated group"
	}
	return ""
}

// findClassEnd returns the index of the ']' that closes the character class
// opening at start. It mirrors findClassEnd() in the TS lib, including the
// literal-member rules for a leading '^' and ']'. If no closing ']' exists it
// returns the last index of the slice (mirroring the TS `p.length - 1`).
func findClassEnd(runes []rune, start int) int {
	n := len(runes)
	i := start + 1
	if i < n && runes[i] == '^' {
		i++
	}
	if i < n && runes[i] == ']' {
		i++
	}
	for i < n && runes[i] != ']' {
		if runes[i] == '\\' {
			i++
		}
		i++
	}
	if i < n {
		return i
	}
	return n - 1
}

// findGroupEnd returns the index of the ')' that closes the group opening at
// start, honoring nested groups, escapes, and character classes. It mirrors
// findGroupEnd() in the TS lib. If the group never closes it returns the last
// index walked (mirroring the TS `i - 1`).
func findGroupEnd(runes []rune, start int) int {
	n := len(runes)
	depth := 1
	i := start + 1
	for i < n && depth > 0 {
		switch runes[i] {
		case '\\':
			i += 2
			continue
		case '[':
			i = findClassEnd(runes, i) + 1
			continue
		case '(':
			depth++
		case ')':
			depth--
		}
		i++
	}
	return i - 1
}

// describeClass renders the inner text of a character class. It mirrors
// describeClass() in the TS lib: an empty inner set is "(empty)", otherwise
// each backslash is doubled for display.
func describeClass(inner string) string {
	if inner == "" {
		return "(empty)"
	}
	return strings.ReplaceAll(inner, "\\", "\\\\")
}

// describeGroup classifies a group token by its opening syntax. It mirrors
// describeGroup() in the TS lib.
func describeGroup(grp string) string {
	switch {
	case strings.HasPrefix(grp, "(?:"):
		return "non-capturing group"
	case strings.HasPrefix(grp, "(?="):
		return "lookahead assertion (positive)"
	case strings.HasPrefix(grp, "(?!"):
		return "lookahead assertion (negative)"
	case strings.HasPrefix(grp, "(?<="):
		return "lookbehind assertion (positive)"
	case strings.HasPrefix(grp, "(?<!"):
		return "lookbehind assertion (negative)"
	default:
		return "capturing group"
	}
}

// ExplainRegex explains a regex pattern and its flags into labeled tokens. It
// is the Go twin of explainRegex() in src/lib/regexExplain.ts and must agree
// with it on every shared vector. Never panics: an invalid pattern yields
// ExplainResult{OK: false, Error: ...} with empty tokens and flags.
func ExplainRegex(pattern, flags string) ExplainResult {
	if msg := validateRegex(pattern, flags); msg != "" {
		return ExplainResult{OK: false, Error: msg}
	}

	runes := []rune(pattern)
	n := len(runes)
	tokens := make([]RegexToken, 0)
	i := 0
	for i < n {
		ch := runes[i]
		switch ch {
		case '^':
			tokens = append(tokens, RegexToken{"^", "start of the string (or line with /m)"})
			i++
		case '$':
			tokens = append(tokens, RegexToken{"$", "end of the string (or line with /m)"})
			i++
		case '.':
			tokens = append(tokens, RegexToken{".", "any character (except newline, unless /s)"})
			i++
		case '|':
			tokens = append(tokens, RegexToken{"|", "OR — alternation between groups"})
			i++
		case '\\':
			next := ""
			if i+1 < n {
				next = string(runes[i+1])
			}
			desc, ok := escapeDesc[next]
			if !ok {
				desc = fmt.Sprintf("an escaped literal \"%s\"", next)
			}
			tokens = append(tokens, RegexToken{"\\" + next, desc})
			i += 2
		case '[':
			end := findClassEnd(runes, i)
			cls := string(runes[i : end+1])
			negated := i+1 < n && runes[i+1] == '^'
			innerStart := i + 1
			if negated {
				innerStart++
			}
			inner := ""
			if innerStart < end {
				inner = string(runes[innerStart:end])
			}
			kind := "of"
			if negated {
				kind = "character NOT in"
			}
			tokens = append(tokens, RegexToken{cls, fmt.Sprintf("match any %s: %s", kind, describeClass(inner))})
			i = end + 1
		case '(':
			end := findGroupEnd(runes, i)
			grp := string(runes[i : end+1])
			tokens = append(tokens, RegexToken{grp, describeGroup(grp)})
			i = end + 1
		case '*', '+', '?':
			lazy := i+1 < n && runes[i+1] == '?'
			base := ""
			switch ch {
			case '*':
				base = "0 or more times"
			case '+':
				base = "1 or more times"
			case '?':
				base = "0 or 1 time (optional)"
			}
			suffix := " (greedy)"
			if lazy {
				suffix = " (lazy/non-greedy)"
			}
			tok := string(ch)
			if lazy {
				tok += "?"
			}
			tokens = append(tokens, RegexToken{tok, "quantifier — " + base + suffix})
			if lazy {
				i += 2
			} else {
				i++
			}
		case '{':
			// Find the closing '}'. When there is one this is a repeat
			// quantifier; otherwise the '{' is a literal (JS only treats a
			// well-formed {n}, {n,}, {n,m} as a quantifier).
			end := -1
			for k := i; k < n; k++ {
				if runes[k] == '}' {
					end = k
					break
				}
			}
			if end != -1 {
				q := string(runes[i : end+1])
				lazy := end+1 < n && runes[end+1] == '?'
				repeat := string(runes[i+1 : end]) // q without the surrounding { }
				suffix := ""
				if lazy {
					suffix = " (lazy)"
				}
				tok := q
				if lazy {
					tok += "?"
				}
				tokens = append(tokens, RegexToken{tok, "quantifier — repeat " + repeat + " time(s)" + suffix})
				i = end + 1
				if lazy {
					i++
				}
			} else {
				tokens = append(tokens, RegexToken{string(ch), "the literal \"" + escapeHtmlish(string(ch)) + "\""})
				i++
			}
		default:
			tokens = append(tokens, RegexToken{string(ch), "the literal \"" + escapeHtmlish(string(ch)) + "\""})
			i++
		}
	}

	flagList := make([]FlagDesc, 0, len(flags))
	for _, f := range flags {
		d, ok := DescribeFlag(f)
		if !ok {
			d = fmt.Sprintf("unknown flag \"%s\"", string(f))
		}
		flagList = append(flagList, FlagDesc{Flag: string(f), Description: d})
	}

	return ExplainResult{OK: true, Tokens: tokens, Flags: flagList}
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →