Skip to content

Regex Explainer — JavaScript source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.

/**
 * Regex Explainer - JavaScript port.
 *
 * Language:    JavaScript (ES2020+: classes via JSDoc, nullish coalescing, optional chaining).
 * Source:      CosmoDev polyglot showcase port of the `regex-explainer` tool.
 * Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
 *
 * What it does:
 *   Tokenizes a regular expression into labeled tokens - anchors, escapes,
 *   character classes, quantifiers, groups, alternation, and literals - and
 *   describes each flag. Never throws: an invalid pattern yields a structured
 *   error result instead.
 *
 * This is display source - part of CosmoDev's polyglot tool pages, where every
 * tool's pure logic is shown side-by-side in multiple languages.
 */

/**
 * @typedef {Object} RegexToken
 * @property {string} token       The raw slice of the pattern this entry covers.
 * @property {string} description A human-readable explanation of that slice.
 */

/**
 * @typedef {Object} FlagInfo
 * @property {string} flag
 * @property {string} description
 */

/**
 * @typedef {Object} ExplainResult
 * @property {boolean} ok        Whether the pattern compiled successfully.
 * @property {RegexToken[]} tokens
 * @property {FlagInfo[]} flags
 * @property {string|null} error  Engine error message when ok is false.
 */

/** Human descriptions for each supported pattern flag. */
const FLAG_DESC = {
  g: 'global - find all matches',
  i: 'case-insensitive',
  m: 'multiline (^ and $ match line boundaries)',
  s: 'dotAll - "." matches newlines',
  u: 'unicode',
  y: 'sticky - match at lastIndex',
  d: 'indices - expose match boundaries',
};

/** Descriptions for backslash escape sequences inside a pattern. */
const ESCAPE_DESC = {
  d: 'a digit [0-9]',
  D: 'a non-digit',
  w: 'a word character [A-Za-z0-9_]',
  W: 'a non-word character',
  s: 'a whitespace character',
  S: 'a non-whitespace character',
  b: 'a word boundary',
  B: 'a non-word boundary',
  n: 'a newline',
  t: 'a tab',
  r: 'a carriage return',
};

/**
 * Escape a character so it renders cleanly inside a double-quoted description.
 * The TS original only neutralizes the double quote; we mirror that exactly.
 */
function escapeHtmlish(s) {
  return s.replace(/"/g, '\\"');
}

/**
 * Look up the description for a single flag letter.
 * @param {string} flag
 * @returns {string|null}  The description, or null if the flag is unrecognized.
 */
export function describeFlag(flag) {
  return FLAG_DESC[flag] ?? null;
}

/**
 * Explain a regex pattern + flags into a flat list of labeled tokens.
 *
 * @param {string} pattern
 * @param {string} [flags='']
 * @returns {ExplainResult}
 */
export function explainRegex(pattern, flags = '') {
  // First, ask the JS regex engine to validate the pattern. This catches
  // malformed input cheaply and gives us a precise error message.
  try {
    // eslint-disable-next-line no-new
    new RegExp(pattern, flags);
  } catch (e) {
    return { ok: false, tokens: [], flags: [], error: e.message };
  }

  /** @type {RegexToken[]} */
  const tokens = [];
  const p = pattern;
  let i = 0;

  /** Append a labeled token. */
  const push = (token, description) => tokens.push({ token, description });

  while (i < p.length) {
    const ch = p[i];

    // --- Anchors / single-char metacharacters ---------------------------
    if (ch === '^') {
      push('^', 'start of the string (or line with /m)');
      i++;
      continue;
    }
    if (ch === '$') {
      push('$', 'end of the string (or line with /m)');
      i++;
      continue;
    }
    if (ch === '.') {
      push('.', 'any character (except newline, unless /s)');
      i++;
      continue;
    }
    if (ch === '|') {
      push('|', 'OR - alternation between groups');
      i++;
      continue;
    }

    // --- Backslash escape sequences --------------------------------------
    if (ch === '\\') {
      const next = p[i + 1] ?? '';
      const desc = ESCAPE_DESC[next] ?? `an escaped literal "${next}"`;
      push('\\' + next, desc);
      i += 2;
      continue;
    }

    // --- Character class [ ... ] -----------------------------------------
    if (ch === '[') {
      const end = findClassEnd(p, i);
      const cls = p.slice(i, end + 1);
      const negated = p[i + 1] === '^';
      const inner = cls.slice(1 + (negated ? 1 : 0), -1);
      push(cls, `match any ${negated ? 'character NOT in' : 'of'}: ${describeClass(inner)}`);
      i = end + 1;
      continue;
    }

    // --- Group ( ... ) - capturing, non-capturing, lookaround ------------
    if (ch === '(') {
      const end = findGroupEnd(p, i);
      const grp = p.slice(i, end + 1);
      push(grp, describeGroup(grp));
      i = end + 1;
      continue;
    }

    // --- Quantifiers that attach to the previous token -------------------
    if (ch === '*' || ch === '+' || ch === '?') {
      const lazy = p[i + 1] === '?';
      const base =
        ch === '*' ? '0 or more times' : ch === '+' ? '1 or more times' : '0 or 1 time (optional)';
      push(ch + (lazy ? '?' : ''), `quantifier - ${base}${lazy ? ' (lazy/non-greedy)' : ' (greedy)'}`);
      i += lazy ? 2 : 1;
      continue;
    }
    if (ch === '{') {
      // Bounded quantifier like {3} or {2,5}. Find its closing brace.
      const end = p.indexOf('}', i);
      if (end !== -1) {
        const q = p.slice(i, end + 1);
        const lazy = p[end + 1] === '?';
        push(q + (lazy ? '?' : ''), `quantifier - repeat ${q.slice(1, -1)} time(s)${lazy ? ' (lazy)' : ''}`);
        i = end + 1 + (lazy ? 1 : 0);
        continue;
      }
      // No closing brace: fall through and treat '{' as a literal.
    }

    // --- Default: a literal character ------------------------------------
    push(ch, `the literal "${escapeHtmlish(ch)}"`);
    i++;
  }

  // Describe each flag, surfacing unknown flags explicitly.
  const flagList = flags.split('').map((f) => ({
    flag: f,
    description: describeFlag(f) ?? `unknown flag "${f}"`,
  }));

  return { ok: true, tokens, flags: flagList, error: null };
}

/**
 * Find the index of the `]` that closes a character class starting at `start`.
 *
 * Inside a class, a leading `]` is treated as a literal, and backslash escapes
 * the next character (so `\]` is a literal close bracket).
 */
function findClassEnd(p, start) {
  let i = start + 1;
  if (p[i] === '^') i++; // negation: [^...]
  if (p[i] === ']') i++; // a leading ] is a literal, not a terminator
  while (i < p.length && p[i] !== ']') {
    if (p[i] === '\\') i++; // skip the escaped character
    i++;
  }
  // If we ran off the end, point at the last char (validation already ran).
  return i < p.length ? i : p.length - 1;
}

/**
 * Find the index of the `)` that closes the group opened at `start`.
 *
 * Tracks nested parentheses, skips over character classes wholesale, and skips
 * backslash-escaped characters so they cannot confuse the depth counter.
 */
function findGroupEnd(p, start) {
  let depth = 1;
  let i = start + 1;
  while (i < p.length && depth > 0) {
    if (p[i] === '\\') {
      i += 2;
      continue;
    }
    if (p[i] === '[') {
      i = findClassEnd(p, i) + 1;
      continue;
    }
    if (p[i] === '(') depth++;
    else if (p[i] === ')') depth--;
    i++;
  }
  return i - 1;
}

/** Render the inside of a character class for display. */
function describeClass(inner) {
  if (!inner) return '(empty)';
  // Double backslashes so they display as a single literal backslash.
  return inner.replace(/\\/g, '\\\\');
}

/** Classify a group by its opening syntax. */
function describeGroup(grp) {
  if (grp.startsWith('(?:')) return 'non-capturing group';
  if (grp.startsWith('(?=')) return 'lookahead assertion (positive)';
  if (grp.startsWith('(?!')) return 'lookahead assertion (negative)';
  if (grp.startsWith('(?<=')) return 'lookbehind assertion (positive)';
  if (grp.startsWith('(?<!')) return 'lookbehind assertion (negative)';
  return 'capturing group';
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →