Regex Explainer — Python source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.
"""Regex Explainer — Python port.
Language: Python 3.9+ (stdlib only: dataclasses, re, typing).
Source: CosmoDev polyglot showcase port of the ``regex-explainer`` tool.
Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
What it does:
Tokenizes a regular expression into labeled tokens — anchors, escapes,
character classes, quantifiers, groups, alternation, and literals — and
describes each flag. Never raises from the explainer itself: an invalid
pattern yields a structured error result.
Engine note:
Validation uses Python's ``re`` module, whose syntax differs slightly from
the JavaScript engine (e.g. named groups use ``(?P<name>)`` rather than
``(?<name>)`` in older Pythons). Patterns valid in JS but not accepted by
Python's ``re`` will report ``ok=False`` here. The tokenizer itself still
recognizes lookaround syntax, so descriptions are produced for any
structurally well-formed pattern.
This is display source — part of CosmoDev's polyglot tool pages.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from typing import List, Optional
@dataclass
class RegexToken:
"""A labeled slice of a regex pattern."""
token: str
description: str
@dataclass
class FlagInfo:
"""A flag letter paired with its human description."""
flag: str
description: str
@dataclass
class ExplainResult:
"""The full output of :func:`explain_regex`."""
ok: bool
tokens: List[RegexToken] = field(default_factory=list)
flags: List[FlagInfo] = field(default_factory=list)
error: Optional[str] = None
# Human descriptions for each supported pattern flag.
FLAG_DESC: dict[str, str] = {
"g": "global — find all matches",
"i": "case-insensitive",
"m": "multiline (^ and $ match line boundaries)",
"s": 'dotAll — "." matches newlines',
"u": "unicode",
"y": "sticky — match at lastIndex",
"d": "indices — expose match boundaries",
}
# Descriptions for backslash escape sequences inside a pattern.
ESCAPE_DESC: dict[str, str] = {
"d": "a digit [0-9]",
"D": "a non-digit",
"w": "a word character [A-Za-z0-9_]",
"W": "a non-word character",
"s": "a whitespace character",
"S": "a non-whitespace character",
"b": "a word boundary",
"B": "a non-word boundary",
"n": "a newline",
"t": "a tab",
"r": "a carriage return",
}
def describe_flag(flag: str) -> Optional[str]:
"""Description for a single flag letter, or None if unrecognized."""
return FLAG_DESC.get(flag)
def _python_flags(flags: str) -> int:
"""Translate JS pattern flags into Python ``re`` flag bits for validation.
Only the flags that affect pattern compilation are forwarded; ``g``, ``y``
and ``d`` are API-level in JS and have no Python ``re`` equivalent.
"""
table = {
"i": re.IGNORECASE,
"m": re.MULTILINE,
"s": re.DOTALL,
# "u" is the default under str patterns in Python 3; no bit needed.
}
bits = 0
for f in flags:
bits |= table.get(f, 0)
return bits
def explain_regex(pattern: str, flags: str = "") -> ExplainResult:
"""Explain a regex pattern + flags into a flat list of labeled tokens.
Returns an :class:`ExplainResult`; never raises.
"""
# Validate with Python's re engine before tokenizing.
try:
re.compile(pattern, _python_flags(flags))
except re.error as exc:
return ExplainResult(ok=False, error=str(exc))
# Python strings index by code point, matching JS's per-character walk.
p = pattern
tokens: List[RegexToken] = []
def push(token: str, description: str) -> None:
tokens.append(RegexToken(token, description))
i = 0
while i < len(p):
ch = p[i]
# --- Anchors / single-char metacharacters ------------------------
if ch == "^":
push("^", "start of the string (or line with /m)")
i += 1
continue
if ch == "$":
push("$", "end of the string (or line with /m)")
i += 1
continue
if ch == ".":
push(".", "any character (except newline, unless /s)")
i += 1
continue
if ch == "|":
push("|", "OR — alternation between groups")
i += 1
continue
# --- Backslash escape sequences ----------------------------------
if ch == "\\":
next_ch = p[i + 1] if i + 1 < len(p) else ""
desc = ESCAPE_DESC.get(next_ch, f'an escaped literal "{next_ch}"')
push("\\" + next_ch, desc)
i += 2
continue
# --- Character class [ ... ] -------------------------------------
if ch == "[":
end = _find_class_end(p, i)
cls = p[i : end + 1]
negated = (i + 1 < len(p)) and p[i + 1] == "^"
inner_start = i + 1 + (1 if negated else 0)
inner = p[inner_start:end]
qualifier = "character NOT in" if negated else "of"
push(cls, f"match any {qualifier}: {_describe_class(inner)}")
i = end + 1
continue
# --- Group ( ... ) — capturing, non-capturing, lookaround ---------
if ch == "(":
end = _find_group_end(p, i)
grp = p[i : end + 1]
push(grp, _describe_group(grp))
i = end + 1
continue
# --- Quantifiers that attach to the previous token ----------------
if ch in ("*", "+", "?"):
lazy = (i + 1 < len(p)) and p[i + 1] == "?"
base = {
"*": "0 or more times",
"+": "1 or more times",
"?": "0 or 1 time (optional)",
}[ch]
token = ch + ("?" if lazy else "")
suffix = " (lazy/non-greedy)" if lazy else " (greedy)"
push(token, f"quantifier — {base}{suffix}")
i += 2 if lazy else 1
continue
if ch == "{":
# Bounded quantifier {n} or {n,m}.
end = p.find("}", i)
if end != -1:
q = p[i : end + 1]
lazy = (end + 1 < len(p)) and p[end + 1] == "?"
token = q + ("?" if lazy else "")
suffix = " (lazy)" if lazy else ""
push(token, f"quantifier — repeat {q[1:-1]} time(s){suffix}")
i = end + 1 + (1 if lazy else 0)
continue
# No closing brace: fall through and treat '{' as a literal.
# --- Default: a literal character --------------------------------
push(ch, f'the literal "{_escape_htmlish(ch)}"')
i += 1
# Describe each flag, surfacing unknown flags explicitly.
flag_list = [
FlagInfo(flag=f, description=describe_flag(f) or f'unknown flag "{f}"')
for f in flags
]
return ExplainResult(ok=True, tokens=tokens, flags=flag_list)
def _find_class_end(p: str, start: int) -> int:
"""Index of the ``]`` closing a character class opened at *start*.
A leading ``]`` (right after ``[`` or ``[^``) counts as a literal member,
and ``\\`` escapes the next character.
"""
i = start + 1
if i < len(p) and p[i] == "^":
i += 1
if i < len(p) and p[i] == "]":
i += 1 # leading ] is a literal, not a terminator
while i < len(p) and p[i] != "]":
if p[i] == "\\":
i += 1 # skip the escaped char
i += 1
return i if i < len(p) else len(p) - 1
def _find_group_end(p: str, start: int) -> int:
"""Index of the ``)`` matching the group opened at *start*.
Tracks nesting depth, skips character classes wholesale, and skips escapes.
"""
depth = 1
i = start + 1
while i < len(p) and depth > 0:
if p[i] == "\\":
i += 2
continue
if p[i] == "[":
i = _find_class_end(p, i) + 1
continue
if p[i] == "(":
depth += 1
elif p[i] == ")":
depth -= 1
i += 1
return i - 1
def _describe_class(inner: str) -> str:
"""Render the inside of a character class for display."""
if not inner:
return "(empty)"
# Double backslashes so they display as a single literal backslash.
return inner.replace("\\", "\\\\")
def _describe_group(grp: str) -> str:
"""Classify a group by its opening syntax."""
if grp.startswith("(?:"):
return "non-capturing group"
if grp.startswith("(?="):
return "lookahead assertion (positive)"
if grp.startswith("(?!"):
return "lookahead assertion (negative)"
if grp.startswith("(?<="):
return "lookbehind assertion (positive)"
if grp.startswith("(?<!"):
return "lookbehind assertion (negative)"
return "capturing group"
def _escape_htmlish(s: str) -> str:
"""Neutralize a double quote so it renders inside a quoted description."""
return s.replace('"', '\\"')
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →