Regex Explainer — PHP source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the PHP implementation — the same logic the interactive tool runs, in a shareable, citable form.
<?php
/**
* Regex Explainer — PHP port.
*
* Language: PHP 8+ (PCRE engine, constructor property promotion, mbstring).
* Source: CosmoDev polyglot showcase port of the `regex-explainer` tool.
* Ported from: src/lib/regexExplain.ts (the canonical, live TypeScript lib).
*
* What it does:
* Tokenizes a regular expression into labeled tokens — anchors, escapes,
* character classes, quantifiers, groups, alternation, and literals — and
* describes each flag. Never throws from the explainer itself: an invalid
* pattern yields a structured error result.
*
* Engine note: PHP uses PCRE, which is close to (but not identical to) the
* JavaScript regex engine. Validation wraps the user pattern in `~...~`
* delimiters and asks PCRE to compile it; only pattern-modifier flags that
* PHP understands (i, m, s, u) are forwarded for validation — the JS-only
* flags g/y/d are API-level and don't affect pattern validity.
*
* This is display source — part of CosmoDev's polyglot tool pages.
*/
declare(strict_types=1);
namespace CosmoDev\RegexExplainer;
/**
* A labeled slice of a regex pattern.
*/
class RegexToken
{
public function __construct(
public readonly string $token,
public readonly string $description,
) {
}
}
/**
* A flag letter paired with its human description.
*/
class FlagInfo
{
public function __construct(
public readonly string $flag,
public readonly string $description,
) {
}
}
/**
* The full output of RegexExplainer::explain().
*/
class ExplainResult
{
/**
* @param RegexToken[] $tokens
* @param FlagInfo[] $flags
*/
public function __construct(
public readonly bool $ok,
public readonly array $tokens,
public readonly array $flags,
public readonly ?string $error,
) {
}
}
/**
* Pure regex tokenizer — no I/O, no side effects.
*/
final class RegexExplainer
{
/** Human descriptions for each supported pattern flag. */
private const FLAG_DESC = [
'g' => 'global — find all matches',
'i' => 'case-insensitive',
'm' => 'multiline (^ and $ match line boundaries)',
's' => 'dotAll — "." matches newlines',
'u' => 'unicode',
'y' => 'sticky — match at lastIndex',
'd' => 'indices — expose match boundaries',
];
/** Descriptions for backslash escape sequences inside a pattern. */
private const ESCAPE_DESC = [
'd' => 'a digit [0-9]',
'D' => 'a non-digit',
'w' => 'a word character [A-Za-z0-9_]',
'W' => 'a non-word character',
's' => 'a whitespace character',
'S' => 'a non-whitespace character',
'b' => 'a word boundary',
'B' => 'a non-word boundary',
'n' => 'a newline',
't' => 'a tab',
'r' => 'a carriage return',
];
/**
* Description for a single flag letter, or null if unrecognized.
* Mirrors the TS `describeFlag` (returns `string | null`).
*/
public static function describeFlag(string $flag): ?string
{
return \array_key_exists($flag, self::FLAG_DESC)
? self::FLAG_DESC[$flag]
: null;
}
/**
* Explain a regex pattern + flags into a flat list of labeled tokens.
*
* @return ExplainResult Always returns; never throws.
*/
public static function explain(string $pattern, string $flags = ''): ExplainResult
{
// Validate with the PCRE engine before tokenizing.
if (($error = self::validatePattern($pattern, $flags)) !== null) {
return new ExplainResult(false, [], [], $error);
}
// Split into a per-character array so we index by code point (mirrors
// JS's per-character walk). The `//u` flag makes it UTF-8 aware.
$chars = preg_split('//u', $pattern, -1, PREG_SPLIT_NO_EMPTY) ?: [];
$tokens = [];
$i = 0;
$len = \count($chars);
while ($i < $len) {
$ch = $chars[$i];
// --- Anchors / single-char metacharacters ----------------------
if ($ch === '^') {
$tokens[] = new RegexToken('^', 'start of the string (or line with /m)');
$i++;
continue;
}
if ($ch === '$') {
$tokens[] = new RegexToken('$', 'end of the string (or line with /m)');
$i++;
continue;
}
if ($ch === '.') {
$tokens[] = new RegexToken('.', 'any character (except newline, unless /s)');
$i++;
continue;
}
if ($ch === '|') {
$tokens[] = new RegexToken('|', 'OR — alternation between groups');
$i++;
continue;
}
// --- Backslash escape sequences --------------------------------
if ($ch === '\\') {
$next = $chars[$i + 1] ?? '';
$desc = self::ESCAPE_DESC[$next]
?? \sprintf('an escaped literal "%s"', $next);
$tokens[] = new RegexToken('\\' . $next, $desc);
$i += 2;
continue;
}
// --- Character class [ ... ] -----------------------------------
if ($ch === '[') {
$end = self::findClassEnd($chars, $i);
$cls = implode('', \array_slice($chars, $i, $end - $i + 1));
$negated = ($chars[$i + 1] ?? '') === '^';
$innerStart = $i + 1 + ($negated ? 1 : 0);
$inner = implode('', \array_slice($chars, $innerStart, $end - $innerStart));
$qualifier = $negated ? 'character NOT in' : 'of';
$tokens[] = new RegexToken(
$cls,
\sprintf('match any %s: %s', $qualifier, self::describeClass($inner)),
);
$i = $end + 1;
continue;
}
// --- Group ( ... ) — capturing, non-capturing, lookaround -------
if ($ch === '(') {
$end = self::findGroupEnd($chars, $i);
$grp = implode('', \array_slice($chars, $i, $end - $i + 1));
$tokens[] = new RegexToken($grp, self::describeGroup($grp));
$i = $end + 1;
continue;
}
// --- Quantifiers that attach to the previous token --------------
if ($ch === '*' || $ch === '+' || $ch === '?') {
$lazy = ($chars[$i + 1] ?? '') === '?';
$base = $ch === '*' ? '0 or more times'
: ($ch === '+' ? '1 or more times' : '0 or 1 time (optional)');
$token = $ch . ($lazy ? '?' : '');
$suffix = $lazy ? ' (lazy/non-greedy)' : ' (greedy)';
$tokens[] = new RegexToken($token, \sprintf('quantifier — %s%s', $base, $suffix));
$i += $lazy ? 2 : 1;
continue;
}
if ($ch === '{') {
// Bounded quantifier {n} or {n,m}.
$end = self::findBraceEnd($chars, $i);
if ($end !== null) {
$q = implode('', \array_slice($chars, $i, $end - $i + 1));
$lazy = ($chars[$end + 1] ?? '') === '?';
$token = $q . ($lazy ? '?' : '');
$suffix = $lazy ? ' (lazy)' : '';
$inner = implode('', \array_slice($chars, $i + 1, $end - ($i + 1)));
$tokens[] = new RegexToken(
$token,
\sprintf('quantifier — repeat %s time(s)%s', $inner, $suffix),
);
$i = $end + 1 + ($lazy ? 1 : 0);
continue;
}
// No closing brace: fall through and treat '{' as a literal.
}
// --- Default: a literal character ------------------------------
$tokens[] = new RegexToken(
$ch,
\sprintf('the literal "%s"', self::escapeHtmlish($ch)),
);
$i++;
}
// Describe each flag, surfacing unknown flags explicitly.
$flagList = [];
foreach (preg_split('//u', $flags, -1, PREG_SPLIT_NO_EMPTY) ?: [] as $f) {
$desc = self::describeFlag($f) ?? \sprintf('unknown flag "%s"', $f);
$flagList[] = new FlagInfo($f, $desc);
}
return new ExplainResult(true, $tokens, $flagList, null);
}
/**
* Validate a pattern by asking PCRE to compile it. Returns null when valid,
* or an error message describing why it failed.
*
* The user pattern is embedded between `~` delimiters; the delimiter is
* backslash-escaped (escape-aware) so a `~` inside the pattern doesn't
* terminate the wrapper early.
*/
private static function validatePattern(string $pattern, string $flags): ?string
{
// Forward only the flags PHP understands as pattern modifiers; g/y/d
// are JS API-level and don't affect compile validity.
$phpFlags = preg_replace('/[^imsu]/', '', $flags);
$wrapped = '~' . self::escapeDelimiter($pattern, '~') . '~' . $phpFlags;
try {
@\preg_match($wrapped, '');
} catch (\ValueError $e) {
// PHP 8 throws on a malformed pattern.
return $e->getMessage();
}
// PHP 7 path: a bad pattern sets a non-zero preg error code.
return \preg_last_error() === \PREG_NO_ERROR ? null : self::pregErrorMessage();
}
/**
* Escape the chosen delimiter within the pattern so it can be safely
* wrapped. Escape-aware: it won't double-escape a delimiter that the
* pattern already escapes.
*/
private static function escapeDelimiter(string $s, string $delim): string
{
// Byte-level scan is safe here: both '\' and the delimiter are ASCII,
// and ASCII bytes never appear inside a multibyte UTF-8 sequence.
$out = '';
$backslashes = 0;
$len = \strlen($s);
for ($i = 0; $i < $len; $i++) {
$c = $s[$i];
if ($c === '\\') {
$out .= $c;
$backslashes++;
continue;
}
$escaped = ($backslashes % 2) === 1;
if ($c === $delim && !$escaped) {
$out .= '\\' . $c; // ensure the delimiter is escaped
} else {
$out .= $c;
}
$backslashes = 0;
}
return $out;
}
/** Best-effort human message for the most recent PCRE error. */
private static function pregErrorMessage(): string
{
if (\function_exists('preg_last_error_msg')) {
return \preg_last_error_msg(); // PHP 8.2+
}
return 'invalid regular expression';
}
/**
* Index of the `]` closing a character class opened at $start.
* A leading `]` (right after `[` or `[^`) is a literal member.
*/
private static function findClassEnd(array $chars, int $start): int
{
$len = \count($chars);
$i = $start + 1;
if ($i < $len && $chars[$i] === '^') {
$i++;
}
if ($i < $len && $chars[$i] === ']') {
$i++; // leading ] is a literal, not a terminator
}
while ($i < $len && $chars[$i] !== ']') {
if ($chars[$i] === '\\') {
$i++; // skip the escaped char
}
$i++;
}
return $i < $len ? $i : $len - 1;
}
/**
* Index of the `)` matching the group opened at $start.
* Tracks nesting, skips character classes wholesale, and skips escapes.
*/
private static function findGroupEnd(array $chars, int $start): int
{
$len = \count($chars);
$depth = 1;
$i = $start + 1;
while ($i < $len && $depth > 0) {
if ($chars[$i] === '\\') {
$i += 2;
continue;
}
if ($chars[$i] === '[') {
$i = self::findClassEnd($chars, $i) + 1;
continue;
}
if ($chars[$i] === '(') {
$depth++;
} elseif ($chars[$i] === ')') {
$depth--;
}
$i++;
}
return $i - 1;
}
/**
* Index of the `}` closing a `{...}` quantifier, or null if none.
* An unmatched `{` is treated as a literal by the caller.
*/
private static function findBraceEnd(array $chars, int $start): ?int
{
$len = \count($chars);
for ($i = $start + 1; $i < $len; $i++) {
if ($chars[$i] === '}') {
return $i;
}
}
return null;
}
/** Render the inside of a character class for display. */
private static function describeClass(string $inner): string
{
if ($inner === '') {
return '(empty)';
}
// Double backslashes so they display as a single literal backslash.
return \str_replace('\\', '\\\\', $inner);
}
/** Classify a group by its opening syntax. */
private static function describeGroup(string $grp): string
{
if (str_starts_with($grp, '(?:')) {
return 'non-capturing group';
}
if (str_starts_with($grp, '(?=')) {
return 'lookahead assertion (positive)';
}
if (str_starts_with($grp, '(?!')) {
return 'lookahead assertion (negative)';
}
if (str_starts_with($grp, '(?<=')) {
return 'lookbehind assertion (positive)';
}
if (str_starts_with($grp, '(?<!')) {
return 'lookbehind assertion (negative)';
}
return 'capturing group';
}
/** Neutralize a double quote so it renders inside a quoted description. */
private static function escapeHtmlish(string $s): string
{
return \str_replace('"', '\\"', $s);
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →