Skip to content

Regex Explainer — Ruby source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Ruby implementation — the same logic the interactive tool runs, in a shareable, citable form.

# regex-explainer — Ruby port: tokenize a regex into labeled tokens + describe JS flags.
# Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with Ruby's
# Regexp (Onigmo) — near-JS syntax; JS-only constructs report ok=false here.

FLAG_DESC = {
  'g' => 'global - find all matches', 'i' => 'case-insensitive',
  'm' => 'multiline (^ and $ match line boundaries)', 's' => 'dotAll - "." matches newlines',
  'u' => 'unicode', 'y' => 'sticky - match at lastIndex', 'd' => 'indices - expose match boundaries'
}.freeze

ESCAPE_DESC = {
  'd' => 'a digit [0-9]', 'D' => 'a non-digit', 'w' => 'a word character [A-Za-z0-9_]',
  'W' => 'a non-word character', 's' => 'a whitespace character', 'S' => 'a non-whitespace character',
  'b' => 'a word boundary', 'B' => 'a non-word boundary', 'n' => 'a newline',
  't' => 'a tab', 'r' => 'a carriage return'
}.freeze

QUANT_BASE = { '*' => '0 or more times', '+' => '1 or more times', '?' => '0 or 1 time (optional)' }.freeze

def describe_group(grp)
  return 'non-capturing group' if grp.start_with?('(?:')
  return 'lookahead assertion (positive)' if grp.start_with?('(?=')
  return 'lookahead assertion (negative)' if grp.start_with?('(?!')
  return 'lookbehind assertion (positive)' if grp.start_with?('(?<=')
  return 'lookbehind assertion (negative)' if grp.start_with?('(?<!')

  'capturing group'
end

# Double backslashes so they display as one literal backslash ('(empty)' for no members).
def describe_class(inner)
  return '(empty)' if inner.nil? || inner.empty?

  inner.gsub('\\', '\\\\\\\\')
end

# Index of the ']' closing a class opened at i; a leading ']' is a literal member.
def find_class_end(p, i)
  i += 1
  i += 1 if p[i] == '^'
  i += 1 if p[i] == ']'
  while i < p.length && p[i] != ']'
    i += 2 if p[i] == '\\'
    i += 1
  end
  i < p.length ? i : p.length - 1
end

# Index of the ')' matching the group opened at start; skips classes + escapes.
def find_group_end(p, start)
  depth = 1
  i = start + 1
  while i < p.length && depth.positive?
    if p[i] == '\\'
      i += 2
    elsif p[i] == '['
      i = find_class_end(p, i) + 1
    else
      depth += 1 if p[i] == '('
      depth -= 1 if p[i] == ')'
      i += 1
    end
  end
  i - 1
end

# JS i/s map onto Onigmo bits (JS /m multiline anchors is Ruby's default behavior).
def ruby_options(flags)
  flags.chars.reduce(0) do |bits, f|
    bits | (f == 'i' ? Regexp::IGNORECASE : f == 's' ? Regexp::MULTILINE : 0)
  end
end

# Explain a regex pattern + flags into tokens. Never raises.
def explain_regex(pattern, flags = '')
  Regexp.new(pattern, ruby_options(flags)) # validate with the native engine first

  tokens = []
  push = ->(token, description) { tokens << { token: token, description: description } }
  i = 0
  while i < pattern.length
    ch = pattern[i]
    case ch
    when '^' then push.('^', 'start of the string (or line with /m)'); i += 1
    when '$' then push.('$', 'end of the string (or line with /m)'); i += 1
    when '.' then push.('.', 'any character (except newline, unless /s)'); i += 1
    when '|' then push.('|', 'OR - alternation between groups'); i += 1
    when '\\'
      nxt = pattern[i + 1] || ''
      push.("\\#{nxt}", ESCAPE_DESC.fetch(nxt, "an escaped literal \"#{nxt}\""))
      i += 2
    when '['
      fin = find_class_end(pattern, i)
      cls = pattern[i..fin]
      negated = pattern[i + 1] == '^'
      inner = cls[(1 + (negated ? 1 : 0))..-2].to_s
      what = negated ? 'character NOT in' : 'of'
      push.(cls, "match any #{what}: #{describe_class(inner)}")
      i = fin + 1
    when '('
      fin = find_group_end(pattern, i)
      push.(pattern[i..fin], describe_group(pattern[i..fin]))
      i = fin + 1
    when '*', '+', '?'
      lazy = pattern[i + 1] == '?'
      push.(lazy ? "#{ch}?" : ch, "quantifier - #{QUANT_BASE[ch]}#{lazy ? ' (lazy/non-greedy)' : ' (greedy)'}")
      i += lazy ? 2 : 1
    when '{'
      fin = pattern.index('}', i)
      if fin # bounded quantifier {n,m}
        lazy = pattern[fin + 1] == '?'
        q = pattern[i..fin]
        push.(lazy ? "#{q}?" : q, "quantifier - repeat #{q[1..-2]} time(s)#{lazy ? ' (lazy)' : ''}")
        i = fin + 1 + (lazy ? 1 : 0)
      else # no closing brace: a literal '{'
        push.(ch, "the literal \"{\"")
        i += 1
      end
    else # a literal character
      push.(ch, "the literal \"#{ch}\"")
      i += 1
    end
  end

  flag_list = flags.chars.map { |f| { flag: f, description: FLAG_DESC.fetch(f, "unknown flag \"#{f}\"") } }
  { ok: true, tokens: tokens, flags: flag_list, error: nil }
rescue RegexpError, ArgumentError => e
  { ok: false, tokens: [], flags: [], error: e.message }
end

if __FILE__ == $PROGRAM_NAME
  result = explain_regex('^(\w+)@([\w.-]+)$', 'gi')
  abort("error: #{result[:error]}") unless result[:ok]
  result[:tokens].each { |t| puts format('%-14s %s', t[:token], t[:description]) }
  result[:flags].each { |f| puts "flag #{f[:flag]}: #{f[:description]}" }
end

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →