Skip to content

Regex Explainer — Python source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Python implementation — the same logic the interactive tool runs, in a shareable, citable form.

"""Regex Explainer — Python port.

Language:    Python 3.9+ (stdlib only: dataclasses, re, typing).
Source:      CosmoDev polyglot showcase port of the ``regex-explainer`` tool.
Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).

What it does:
    Tokenizes a regular expression into labeled tokens — anchors, escapes,
    character classes, quantifiers, groups, alternation, and literals — and
    describes each flag. Never raises from the explainer itself: an invalid
    pattern yields a structured error result.

Engine note:
    Validation uses Python's ``re`` module, whose syntax differs slightly from
    the JavaScript engine (e.g. named groups use ``(?P<name>)`` rather than
    ``(?<name>)`` in older Pythons). Patterns valid in JS but not accepted by
    Python's ``re`` will report ``ok=False`` here. The tokenizer itself still
    recognizes lookaround syntax, so descriptions are produced for any
    structurally well-formed pattern.

This is display source — part of CosmoDev's polyglot tool pages.
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field
from typing import List, Optional


@dataclass
class RegexToken:
    """A labeled slice of a regex pattern."""

    token: str
    description: str


@dataclass
class FlagInfo:
    """A flag letter paired with its human description."""

    flag: str
    description: str


@dataclass
class ExplainResult:
    """The full output of :func:`explain_regex`."""

    ok: bool
    tokens: List[RegexToken] = field(default_factory=list)
    flags: List[FlagInfo] = field(default_factory=list)
    error: Optional[str] = None


# Human descriptions for each supported pattern flag.
FLAG_DESC: dict[str, str] = {
    "g": "global — find all matches",
    "i": "case-insensitive",
    "m": "multiline (^ and $ match line boundaries)",
    "s": 'dotAll — "." matches newlines',
    "u": "unicode",
    "y": "sticky — match at lastIndex",
    "d": "indices — expose match boundaries",
}

# Descriptions for backslash escape sequences inside a pattern.
ESCAPE_DESC: dict[str, str] = {
    "d": "a digit [0-9]",
    "D": "a non-digit",
    "w": "a word character [A-Za-z0-9_]",
    "W": "a non-word character",
    "s": "a whitespace character",
    "S": "a non-whitespace character",
    "b": "a word boundary",
    "B": "a non-word boundary",
    "n": "a newline",
    "t": "a tab",
    "r": "a carriage return",
}


def describe_flag(flag: str) -> Optional[str]:
    """Description for a single flag letter, or None if unrecognized."""
    return FLAG_DESC.get(flag)


def _python_flags(flags: str) -> int:
    """Translate JS pattern flags into Python ``re`` flag bits for validation.

    Only the flags that affect pattern compilation are forwarded; ``g``, ``y``
    and ``d`` are API-level in JS and have no Python ``re`` equivalent.
    """
    table = {
        "i": re.IGNORECASE,
        "m": re.MULTILINE,
        "s": re.DOTALL,
        # "u" is the default under str patterns in Python 3; no bit needed.
    }
    bits = 0
    for f in flags:
        bits |= table.get(f, 0)
    return bits


def explain_regex(pattern: str, flags: str = "") -> ExplainResult:
    """Explain a regex pattern + flags into a flat list of labeled tokens.

    Returns an :class:`ExplainResult`; never raises.
    """
    # Validate with Python's re engine before tokenizing.
    try:
        re.compile(pattern, _python_flags(flags))
    except re.error as exc:
        return ExplainResult(ok=False, error=str(exc))

    # Python strings index by code point, matching JS's per-character walk.
    p = pattern
    tokens: List[RegexToken] = []

    def push(token: str, description: str) -> None:
        tokens.append(RegexToken(token, description))

    i = 0
    while i < len(p):
        ch = p[i]

        # --- Anchors / single-char metacharacters ------------------------
        if ch == "^":
            push("^", "start of the string (or line with /m)")
            i += 1
            continue
        if ch == "$":
            push("$", "end of the string (or line with /m)")
            i += 1
            continue
        if ch == ".":
            push(".", "any character (except newline, unless /s)")
            i += 1
            continue
        if ch == "|":
            push("|", "OR — alternation between groups")
            i += 1
            continue

        # --- Backslash escape sequences ----------------------------------
        if ch == "\\":
            next_ch = p[i + 1] if i + 1 < len(p) else ""
            desc = ESCAPE_DESC.get(next_ch, f'an escaped literal "{next_ch}"')
            push("\\" + next_ch, desc)
            i += 2
            continue

        # --- Character class [ ... ] -------------------------------------
        if ch == "[":
            end = _find_class_end(p, i)
            cls = p[i : end + 1]
            negated = (i + 1 < len(p)) and p[i + 1] == "^"
            inner_start = i + 1 + (1 if negated else 0)
            inner = p[inner_start:end]
            qualifier = "character NOT in" if negated else "of"
            push(cls, f"match any {qualifier}: {_describe_class(inner)}")
            i = end + 1
            continue

        # --- Group ( ... ) — capturing, non-capturing, lookaround ---------
        if ch == "(":
            end = _find_group_end(p, i)
            grp = p[i : end + 1]
            push(grp, _describe_group(grp))
            i = end + 1
            continue

        # --- Quantifiers that attach to the previous token ----------------
        if ch in ("*", "+", "?"):
            lazy = (i + 1 < len(p)) and p[i + 1] == "?"
            base = {
                "*": "0 or more times",
                "+": "1 or more times",
                "?": "0 or 1 time (optional)",
            }[ch]
            token = ch + ("?" if lazy else "")
            suffix = " (lazy/non-greedy)" if lazy else " (greedy)"
            push(token, f"quantifier — {base}{suffix}")
            i += 2 if lazy else 1
            continue
        if ch == "{":
            # Bounded quantifier {n} or {n,m}.
            end = p.find("}", i)
            if end != -1:
                q = p[i : end + 1]
                lazy = (end + 1 < len(p)) and p[end + 1] == "?"
                token = q + ("?" if lazy else "")
                suffix = " (lazy)" if lazy else ""
                push(token, f"quantifier — repeat {q[1:-1]} time(s){suffix}")
                i = end + 1 + (1 if lazy else 0)
                continue
            # No closing brace: fall through and treat '{' as a literal.

        # --- Default: a literal character --------------------------------
        push(ch, f'the literal "{_escape_htmlish(ch)}"')
        i += 1

    # Describe each flag, surfacing unknown flags explicitly.
    flag_list = [
        FlagInfo(flag=f, description=describe_flag(f) or f'unknown flag "{f}"')
        for f in flags
    ]

    return ExplainResult(ok=True, tokens=tokens, flags=flag_list)


def _find_class_end(p: str, start: int) -> int:
    """Index of the ``]`` closing a character class opened at *start*.

    A leading ``]`` (right after ``[`` or ``[^``) counts as a literal member,
    and ``\\`` escapes the next character.
    """
    i = start + 1
    if i < len(p) and p[i] == "^":
        i += 1
    if i < len(p) and p[i] == "]":
        i += 1  # leading ] is a literal, not a terminator
    while i < len(p) and p[i] != "]":
        if p[i] == "\\":
            i += 1  # skip the escaped char
        i += 1
    return i if i < len(p) else len(p) - 1


def _find_group_end(p: str, start: int) -> int:
    """Index of the ``)`` matching the group opened at *start*.

    Tracks nesting depth, skips character classes wholesale, and skips escapes.
    """
    depth = 1
    i = start + 1
    while i < len(p) and depth > 0:
        if p[i] == "\\":
            i += 2
            continue
        if p[i] == "[":
            i = _find_class_end(p, i) + 1
            continue
        if p[i] == "(":
            depth += 1
        elif p[i] == ")":
            depth -= 1
        i += 1
    return i - 1


def _describe_class(inner: str) -> str:
    """Render the inside of a character class for display."""
    if not inner:
        return "(empty)"
    # Double backslashes so they display as a single literal backslash.
    return inner.replace("\\", "\\\\")


def _describe_group(grp: str) -> str:
    """Classify a group by its opening syntax."""
    if grp.startswith("(?:"):
        return "non-capturing group"
    if grp.startswith("(?="):
        return "lookahead assertion (positive)"
    if grp.startswith("(?!"):
        return "lookahead assertion (negative)"
    if grp.startswith("(?<="):
        return "lookbehind assertion (positive)"
    if grp.startswith("(?<!"):
        return "lookbehind assertion (negative)"
    return "capturing group"


def _escape_htmlish(s: str) -> str:
    """Neutralize a double quote so it renders inside a quoted description."""
    return s.replace('"', '\\"')

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →