Skip to content

Regex Explainer — TypeScript source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the TypeScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

// Pure regex explainer - no React, no DOM, deterministic.
// Tokenizes a regular expression into labeled tokens (anchors, escapes,
// classes, quantifiers, groups, alternation, literals) and describes flags.
// Never throws.

export interface RegexToken {
  token: string;
  description: string;
}

export interface ExplainResult {
  ok: boolean;
  tokens: RegexToken[];
  flags: { flag: string; description: string }[];
  error: string | null;
}

const FLAG_DESC: Record<string, string> = {
  g: 'global - find all matches',
  i: 'case-insensitive',
  m: 'multiline (^ and $ match line boundaries)',
  s: 'dotAll - "." matches newlines',
  u: 'unicode',
  y: 'sticky - match at lastIndex',
  d: 'indices - expose match boundaries',
};

export function describeFlag(flag: string): string | null {
  return FLAG_DESC[flag] ?? null;
}

const ESCAPE_DESC: Record<string, string> = {
  d: 'a digit [0-9]',
  D: 'a non-digit',
  w: 'a word character [A-Za-z0-9_]',
  W: 'a non-word character',
  s: 'a whitespace character',
  S: 'a non-whitespace character',
  b: 'a word boundary',
  B: 'a non-word boundary',
  n: 'a newline',
  t: 'a tab',
  r: 'a carriage return',
};

function escapeHtmlish(s: string): string {
  return s.replace(/"/g, '\\"');
}

/** Explain a regex pattern + flags into tokens. Never throws. */
export function explainRegex(pattern: string, flags = ''): ExplainResult {
  try {
    // eslint-disable-next-line no-new
    new RegExp(pattern, flags);
  } catch (e) {
    return { ok: false, tokens: [], flags: [], error: (e as Error).message };
  }

  const tokens: RegexToken[] = [];
  let i = 0;
  const p = pattern;

  const push = (token: string, description: string) => tokens.push({ token, description });

  while (i < p.length) {
    const ch = p[i];

    if (ch === '^') {
      push('^', 'start of the string (or line with /m)');
      i++;
      continue;
    }
    if (ch === '$') {
      push('$', 'end of the string (or line with /m)');
      i++;
      continue;
    }
    if (ch === '.') {
      push('.', 'any character (except newline, unless /s)');
      i++;
      continue;
    }
    if (ch === '|') {
      push('|', 'OR - alternation between groups');
      i++;
      continue;
    }
    if (ch === '\\') {
      const next = p[i + 1] ?? '';
      const desc = ESCAPE_DESC[next] ?? `an escaped literal "${next}"`;
      push('\\' + next, desc);
      i += 2;
      continue;
    }
    if (ch === '[') {
      const end = findClassEnd(p, i);
      const cls = p.slice(i, end + 1);
      const negated = p[i + 1] === '^';
      const inner = cls.slice(1 + (negated ? 1 : 0), -1);
      push(cls, `match any ${negated ? 'character NOT in' : 'of'}: ${describeClass(inner)}`);
      i = end + 1;
      continue;
    }
    if (ch === '(') {
      const end = findGroupEnd(p, i);
      const grp = p.slice(i, end + 1);
      push(grp, describeGroup(grp));
      i = end + 1;
      continue;
    }
    // Quantifiers attach to the previous token.
    if (ch === '*' || ch === '+' || ch === '?') {
      const lazy = p[i + 1] === '?';
      const base = ch === '*' ? '0 or more times' : ch === '+' ? '1 or more times' : '0 or 1 time (optional)';
      push(ch + (lazy ? '?' : ''), `quantifier - ${base}${lazy ? ' (lazy/non-greedy)' : ' (greedy)'}`);
      i += lazy ? 2 : 1;
      continue;
    }
    if (ch === '{') {
      const end = p.indexOf('}', i);
      if (end !== -1) {
        const q = p.slice(i, end + 1);
        const lazy = p[end + 1] === '?';
        push(q + (lazy ? '?' : ''), `quantifier - repeat ${q.slice(1, -1)} time(s)${lazy ? ' (lazy)' : ''}`);
        i = end + 1 + (lazy ? 1 : 0);
        continue;
      }
    }
    // Default: literal character.
    push(ch, `the literal "${escapeHtmlish(ch)}"`);
    i++;
  }

  const flagList = flags.split('').map((f) => ({ flag: f, description: describeFlag(f) ?? `unknown flag "${f}"` }));

  return { ok: true, tokens, flags: flagList, error: null };
}

function findClassEnd(p: string, start: number): number {
  let i = start + 1;
  if (p[i] === '^') i++;
  if (p[i] === ']') i++; // leading ] is literal
  while (i < p.length && p[i] !== ']') {
    if (p[i] === '\\') i++;
    i++;
  }
  return i < p.length ? i : p.length - 1;
}

function findGroupEnd(p: string, start: number): number {
  let depth = 1;
  let i = start + 1;
  while (i < p.length && depth > 0) {
    if (p[i] === '\\') {
      i += 2;
      continue;
    }
    if (p[i] === '[') {
      i = findClassEnd(p, i) + 1;
      continue;
    }
    if (p[i] === '(') depth++;
    else if (p[i] === ')') depth--;
    i++;
  }
  return i - 1;
}

function describeClass(inner: string): string {
  if (!inner) return '(empty)';
  return inner.replace(/\\/g, '\\\\');
}

function describeGroup(grp: string): string {
  if (grp.startsWith('(?:')) return 'non-capturing group';
  if (grp.startsWith('(?=')) return 'lookahead assertion (positive)';
  if (grp.startsWith('(?!')) return 'lookahead assertion (negative)';
  if (grp.startsWith('(?<=')) return 'lookbehind assertion (positive)';
  if (grp.startsWith('(?<!')) return 'lookbehind assertion (negative)';
  return 'capturing group';
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →