Regex Explainer — Ruby source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.
# regex-explainer — Ruby port: tokenize a regex into labeled tokens + describe JS flags.
# Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with Ruby's
# Regexp (Onigmo) — near-JS syntax; JS-only constructs report ok=false here.
FLAG_DESC = {
'g' => 'global - find all matches', 'i' => 'case-insensitive',
'm' => 'multiline (^ and $ match line boundaries)', 's' => 'dotAll - "." matches newlines',
'u' => 'unicode', 'y' => 'sticky - match at lastIndex', 'd' => 'indices - expose match boundaries'
}.freeze
ESCAPE_DESC = {
'd' => 'a digit [0-9]', 'D' => 'a non-digit', 'w' => 'a word character [A-Za-z0-9_]',
'W' => 'a non-word character', 's' => 'a whitespace character', 'S' => 'a non-whitespace character',
'b' => 'a word boundary', 'B' => 'a non-word boundary', 'n' => 'a newline',
't' => 'a tab', 'r' => 'a carriage return'
}.freeze
QUANT_BASE = { '*' => '0 or more times', '+' => '1 or more times', '?' => '0 or 1 time (optional)' }.freeze
def describe_group(grp)
return 'non-capturing group' if grp.start_with?('(?:')
return 'lookahead assertion (positive)' if grp.start_with?('(?=')
return 'lookahead assertion (negative)' if grp.start_with?('(?!')
return 'lookbehind assertion (positive)' if grp.start_with?('(?<=')
return 'lookbehind assertion (negative)' if grp.start_with?('(?<!')
'capturing group'
end
# Double backslashes so they display as one literal backslash ('(empty)' for no members).
def describe_class(inner)
return '(empty)' if inner.nil? || inner.empty?
inner.gsub('\\', '\\\\\\\\')
end
# Index of the ']' closing a class opened at i; a leading ']' is a literal member.
def find_class_end(p, i)
i += 1
i += 1 if p[i] == '^'
i += 1 if p[i] == ']'
while i < p.length && p[i] != ']'
i += 2 if p[i] == '\\'
i += 1
end
i < p.length ? i : p.length - 1
end
# Index of the ')' matching the group opened at start; skips classes + escapes.
def find_group_end(p, start)
depth = 1
i = start + 1
while i < p.length && depth.positive?
if p[i] == '\\'
i += 2
elsif p[i] == '['
i = find_class_end(p, i) + 1
else
depth += 1 if p[i] == '('
depth -= 1 if p[i] == ')'
i += 1
end
end
i - 1
end
# JS i/s map onto Onigmo bits (JS /m multiline anchors is Ruby's default behavior).
def ruby_options(flags)
flags.chars.reduce(0) do |bits, f|
bits | (f == 'i' ? Regexp::IGNORECASE : f == 's' ? Regexp::MULTILINE : 0)
end
end
# Explain a regex pattern + flags into tokens. Never raises.
def explain_regex(pattern, flags = '')
Regexp.new(pattern, ruby_options(flags)) # validate with the native engine first
tokens = []
push = ->(token, description) { tokens << { token: token, description: description } }
i = 0
while i < pattern.length
ch = pattern[i]
case ch
when '^' then push.('^', 'start of the string (or line with /m)'); i += 1
when '$' then push.('$', 'end of the string (or line with /m)'); i += 1
when '.' then push.('.', 'any character (except newline, unless /s)'); i += 1
when '|' then push.('|', 'OR - alternation between groups'); i += 1
when '\\'
nxt = pattern[i + 1] || ''
push.("\\#{nxt}", ESCAPE_DESC.fetch(nxt, "an escaped literal \"#{nxt}\""))
i += 2
when '['
fin = find_class_end(pattern, i)
cls = pattern[i..fin]
negated = pattern[i + 1] == '^'
inner = cls[(1 + (negated ? 1 : 0))..-2].to_s
what = negated ? 'character NOT in' : 'of'
push.(cls, "match any #{what}: #{describe_class(inner)}")
i = fin + 1
when '('
fin = find_group_end(pattern, i)
push.(pattern[i..fin], describe_group(pattern[i..fin]))
i = fin + 1
when '*', '+', '?'
lazy = pattern[i + 1] == '?'
push.(lazy ? "#{ch}?" : ch, "quantifier - #{QUANT_BASE[ch]}#{lazy ? ' (lazy/non-greedy)' : ' (greedy)'}")
i += lazy ? 2 : 1
when '{'
fin = pattern.index('}', i)
if fin # bounded quantifier {n,m}
lazy = pattern[fin + 1] == '?'
q = pattern[i..fin]
push.(lazy ? "#{q}?" : q, "quantifier - repeat #{q[1..-2]} time(s)#{lazy ? ' (lazy)' : ''}")
i = fin + 1 + (lazy ? 1 : 0)
else # no closing brace: a literal '{'
push.(ch, "the literal \"{\"")
i += 1
end
else # a literal character
push.(ch, "the literal \"#{ch}\"")
i += 1
end
end
flag_list = flags.chars.map { |f| { flag: f, description: FLAG_DESC.fetch(f, "unknown flag \"#{f}\"") } }
{ ok: true, tokens: tokens, flags: flag_list, error: nil }
rescue RegexpError, ArgumentError => e
{ ok: false, tokens: [], flags: [], error: e.message }
end
if __FILE__ == $PROGRAM_NAME
result = explain_regex('^(\w+)@([\w.-]+)$', 'gi')
abort("error: #{result[:error]}") unless result[:ok]
result[:tokens].each { |t| puts format('%-14s %s', t[:token], t[:description]) }
result[:flags].each { |f| puts "flag #{f[:flag]}: #{f[:description]}" }
end
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →