Regex Explainer — JavaScript source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the JavaScript implementation — the same logic the interactive tool runs, in a shareable, citable form.
/**
* Regex Explainer - JavaScript port.
*
* Language: JavaScript (ES2020+: classes via JSDoc, nullish coalescing, optional chaining).
* Source: CosmoDev polyglot showcase port of the `regex-explainer` tool.
* Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
*
* What it does:
* Tokenizes a regular expression into labeled tokens - anchors, escapes,
* character classes, quantifiers, groups, alternation, and literals - and
* describes each flag. Never throws: an invalid pattern yields a structured
* error result instead.
*
* This is display source - part of CosmoDev's polyglot tool pages, where every
* tool's pure logic is shown side-by-side in multiple languages.
*/
/**
* @typedef {Object} RegexToken
* @property {string} token The raw slice of the pattern this entry covers.
* @property {string} description A human-readable explanation of that slice.
*/
/**
* @typedef {Object} FlagInfo
* @property {string} flag
* @property {string} description
*/
/**
* @typedef {Object} ExplainResult
* @property {boolean} ok Whether the pattern compiled successfully.
* @property {RegexToken[]} tokens
* @property {FlagInfo[]} flags
* @property {string|null} error Engine error message when ok is false.
*/
/** Human descriptions for each supported pattern flag. */
const FLAG_DESC = {
g: 'global - find all matches',
i: 'case-insensitive',
m: 'multiline (^ and $ match line boundaries)',
s: 'dotAll - "." matches newlines',
u: 'unicode',
y: 'sticky - match at lastIndex',
d: 'indices - expose match boundaries',
};
/** Descriptions for backslash escape sequences inside a pattern. */
const ESCAPE_DESC = {
d: 'a digit [0-9]',
D: 'a non-digit',
w: 'a word character [A-Za-z0-9_]',
W: 'a non-word character',
s: 'a whitespace character',
S: 'a non-whitespace character',
b: 'a word boundary',
B: 'a non-word boundary',
n: 'a newline',
t: 'a tab',
r: 'a carriage return',
};
/**
* Escape a character so it renders cleanly inside a double-quoted description.
* The TS original only neutralizes the double quote; we mirror that exactly.
*/
function escapeHtmlish(s) {
return s.replace(/"/g, '\\"');
}
/**
* Look up the description for a single flag letter.
* @param {string} flag
* @returns {string|null} The description, or null if the flag is unrecognized.
*/
export function describeFlag(flag) {
return FLAG_DESC[flag] ?? null;
}
/**
* Explain a regex pattern + flags into a flat list of labeled tokens.
*
* @param {string} pattern
* @param {string} [flags='']
* @returns {ExplainResult}
*/
export function explainRegex(pattern, flags = '') {
// First, ask the JS regex engine to validate the pattern. This catches
// malformed input cheaply and gives us a precise error message.
try {
// eslint-disable-next-line no-new
new RegExp(pattern, flags);
} catch (e) {
return { ok: false, tokens: [], flags: [], error: e.message };
}
/** @type {RegexToken[]} */
const tokens = [];
const p = pattern;
let i = 0;
/** Append a labeled token. */
const push = (token, description) => tokens.push({ token, description });
while (i < p.length) {
const ch = p[i];
// --- Anchors / single-char metacharacters ---------------------------
if (ch === '^') {
push('^', 'start of the string (or line with /m)');
i++;
continue;
}
if (ch === '$') {
push('$', 'end of the string (or line with /m)');
i++;
continue;
}
if (ch === '.') {
push('.', 'any character (except newline, unless /s)');
i++;
continue;
}
if (ch === '|') {
push('|', 'OR - alternation between groups');
i++;
continue;
}
// --- Backslash escape sequences --------------------------------------
if (ch === '\\') {
const next = p[i + 1] ?? '';
const desc = ESCAPE_DESC[next] ?? `an escaped literal "${next}"`;
push('\\' + next, desc);
i += 2;
continue;
}
// --- Character class [ ... ] -----------------------------------------
if (ch === '[') {
const end = findClassEnd(p, i);
const cls = p.slice(i, end + 1);
const negated = p[i + 1] === '^';
const inner = cls.slice(1 + (negated ? 1 : 0), -1);
push(cls, `match any ${negated ? 'character NOT in' : 'of'}: ${describeClass(inner)}`);
i = end + 1;
continue;
}
// --- Group ( ... ) - capturing, non-capturing, lookaround ------------
if (ch === '(') {
const end = findGroupEnd(p, i);
const grp = p.slice(i, end + 1);
push(grp, describeGroup(grp));
i = end + 1;
continue;
}
// --- Quantifiers that attach to the previous token -------------------
if (ch === '*' || ch === '+' || ch === '?') {
const lazy = p[i + 1] === '?';
const base =
ch === '*' ? '0 or more times' : ch === '+' ? '1 or more times' : '0 or 1 time (optional)';
push(ch + (lazy ? '?' : ''), `quantifier - ${base}${lazy ? ' (lazy/non-greedy)' : ' (greedy)'}`);
i += lazy ? 2 : 1;
continue;
}
if (ch === '{') {
// Bounded quantifier like {3} or {2,5}. Find its closing brace.
const end = p.indexOf('}', i);
if (end !== -1) {
const q = p.slice(i, end + 1);
const lazy = p[end + 1] === '?';
push(q + (lazy ? '?' : ''), `quantifier - repeat ${q.slice(1, -1)} time(s)${lazy ? ' (lazy)' : ''}`);
i = end + 1 + (lazy ? 1 : 0);
continue;
}
// No closing brace: fall through and treat '{' as a literal.
}
// --- Default: a literal character ------------------------------------
push(ch, `the literal "${escapeHtmlish(ch)}"`);
i++;
}
// Describe each flag, surfacing unknown flags explicitly.
const flagList = flags.split('').map((f) => ({
flag: f,
description: describeFlag(f) ?? `unknown flag "${f}"`,
}));
return { ok: true, tokens, flags: flagList, error: null };
}
/**
* Find the index of the `]` that closes a character class starting at `start`.
*
* Inside a class, a leading `]` is treated as a literal, and backslash escapes
* the next character (so `\]` is a literal close bracket).
*/
function findClassEnd(p, start) {
let i = start + 1;
if (p[i] === '^') i++; // negation: [^...]
if (p[i] === ']') i++; // a leading ] is a literal, not a terminator
while (i < p.length && p[i] !== ']') {
if (p[i] === '\\') i++; // skip the escaped character
i++;
}
// If we ran off the end, point at the last char (validation already ran).
return i < p.length ? i : p.length - 1;
}
/**
* Find the index of the `)` that closes the group opened at `start`.
*
* Tracks nested parentheses, skips over character classes wholesale, and skips
* backslash-escaped characters so they cannot confuse the depth counter.
*/
function findGroupEnd(p, start) {
let depth = 1;
let i = start + 1;
while (i < p.length && depth > 0) {
if (p[i] === '\\') {
i += 2;
continue;
}
if (p[i] === '[') {
i = findClassEnd(p, i) + 1;
continue;
}
if (p[i] === '(') depth++;
else if (p[i] === ')') depth--;
i++;
}
return i - 1;
}
/** Render the inside of a character class for display. */
function describeClass(inner) {
if (!inner) return '(empty)';
// Double backslashes so they display as a single literal backslash.
return inner.replace(/\\/g, '\\\\');
}
/** Classify a group by its opening syntax. */
function describeGroup(grp) {
if (grp.startsWith('(?:')) return 'non-capturing group';
if (grp.startsWith('(?=')) return 'lookahead assertion (positive)';
if (grp.startsWith('(?!')) return 'lookahead assertion (negative)';
if (grp.startsWith('(?<=')) return 'lookbehind assertion (positive)';
if (grp.startsWith('(?<!')) return 'lookbehind assertion (negative)';
return 'capturing group';
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →