Regex Explainer — C++ source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.
// regex-explainer — C++ port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with std::regex
// (ECMAScript grammar — the closest std option to the JS engine; JS /s dotAll has no
// std::regex equivalent, and lookbehind is rejected, so such patterns report ok=false).
#include <cstdio>
#include <map>
#include <regex>
#include <string>
#include <utility>
#include <vector>
struct Token { std::string token, description; };
struct Result {
bool ok = false;
std::vector<Token> tokens;
std::vector<std::pair<char, std::string>> flags;
std::string error;
};
static const std::map<std::string, std::string> kFlagDesc = {
{"g", "global - find all matches"}, {"i", "case-insensitive"},
{"m", "multiline (^ and $ match line boundaries)"}, {"s", "dotAll - \".\" matches newlines"},
{"u", "unicode"}, {"y", "sticky - match at lastIndex"}, {"d", "indices - expose match boundaries"}};
static const std::map<std::string, std::string> kEscapeDesc = {
{"d", "a digit [0-9]"}, {"D", "a non-digit"}, {"w", "a word character [A-Za-z0-9_]"},
{"W", "a non-word character"}, {"s", "a whitespace character"}, {"S", "a non-whitespace character"},
{"b", "a word boundary"}, {"B", "a non-word boundary"}, {"n", "a newline"},
{"t", "a tab"}, {"r", "a carriage return"}};
// Index of the ']' closing a class opened at start; a leading ']' is a literal member.
static int findClassEnd(const std::string& p, int start) {
int i = start + 1;
if (i < (int)p.size() && p[i] == '^') i++;
if (i < (int)p.size() && p[i] == ']') i++;
while (i < (int)p.size() && p[i] != ']') { if (p[i] == '\\') i++; i++; }
return i < (int)p.size() ? i : (int)p.size() - 1;
}
// Index of the ')' matching the group opened at start; skips classes + escapes.
static int findGroupEnd(const std::string& p, int start) {
int depth = 1, i = start + 1;
while (i < (int)p.size() && depth > 0) {
if (p[i] == '\\') { i += 2; continue; }
if (p[i] == '[') { i = findClassEnd(p, i) + 1; continue; }
if (p[i] == '(') depth++;
else if (p[i] == ')') depth--;
i++;
}
return i - 1;
}
static std::string describeGroup(const std::string& grp) {
for (auto& [pre, label] : {std::pair{"(?:", "non-capturing group"},
{"(?=", "lookahead assertion (positive)"}, {"(?!", "lookahead assertion (negative)"},
{"(?<=", "lookbehind assertion (positive)"}, {"(?<!", "lookbehind assertion (negative)"}})
if (grp.rfind(pre, 0) == 0) return label; // portable starts_with
return "capturing group";
}
static std::string describeClass(const std::string& inner) {
if (inner.empty()) return "(empty)";
std::string out;
for (char c : inner) { if (c == '\\') out += "\\\\"; out += c; } // double '\' for display
return out;
}
static Result explainRegex(const std::string& pattern, const std::string& flags) {
auto opts = std::regex::ECMAScript; // JS i/m map onto std bits; /s has no equivalent
if (flags.find('i') != std::string::npos) opts |= std::regex::icase;
if (flags.find('m') != std::string::npos) opts |= std::regex::multiline;
try {
std::regex re(pattern, opts); // validate with the native engine first
} catch (const std::regex_error& e) {
return Result{false, {}, {}, e.what()};
}
Result r{true, {}, {}, {}};
auto push = [&](std::string tok, std::string desc) {
r.tokens.push_back(Token{std::move(tok), std::move(desc)});
};
int i = 0;
while (i < (int)pattern.size()) {
char ch = pattern[i];
switch (ch) {
case '^': push("^", "start of the string (or line with /m)"); i++; break;
case '$': push("$", "end of the string (or line with /m)"); i++; break;
case '.': push(".", "any character (except newline, unless /s)"); i++; break;
case '|': push("|", "OR - alternation between groups"); i++; break;
case '\\': {
std::string nxt = i + 1 < (int)pattern.size() ? std::string(1, pattern[i + 1]) : "";
auto it = kEscapeDesc.find(nxt);
push("\\" + nxt, it != kEscapeDesc.end() ? it->second
: "an escaped literal \"" + nxt + "\"");
i += 2;
break;
}
case '[': {
int end = findClassEnd(pattern, i);
std::string cls = pattern.substr(i, end - i + 1);
bool negated = pattern[i + 1] == '^';
std::string inner = cls.substr(1 + (negated ? 1 : 0),
cls.size() - 2 - (negated ? 1 : 0));
push(cls, "match any " + std::string(negated ? "character NOT in" : "of") + ": " +
describeClass(inner));
i = end + 1;
break;
}
case '(': {
int end = findGroupEnd(pattern, i);
std::string grp = pattern.substr(i, end - i + 1);
push(grp, describeGroup(grp));
i = end + 1;
break;
}
case '*': case '+': case '?': {
bool lazy = pattern[i + 1] == '?';
const char* base = ch == '*' ? "0 or more times"
: ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
push(std::string(1, ch) + (lazy ? "?" : ""),
std::string("quantifier - ") + base + (lazy ? " (lazy/non-greedy)" : " (greedy)"));
i += lazy ? 2 : 1;
break;
}
case '{': {
size_t close = pattern.find('}', i);
if (close != std::string::npos) { // bounded quantifier {n,m}
bool lazy = close + 1 < pattern.size() && pattern[close + 1] == '?';
std::string q = pattern.substr(i, close - i + 1);
push(q + (lazy ? "?" : ""), "quantifier - repeat " + q.substr(1, q.size() - 2) +
" time(s)" + (lazy ? " (lazy)" : ""));
i = (int)close + 1 + (lazy ? 1 : 0);
} else { // no closing brace: a literal '{'
push("{", "the literal \"{\"");
i++;
}
break;
}
default:
push(std::string(1, ch), "the literal \"" + std::string(1, ch) + "\"");
i++;
}
}
for (char f : flags)
r.flags.emplace_back(f, kFlagDesc.count(std::string(1, f))
? kFlagDesc.at(std::string(1, f))
: "unknown flag \"" + std::string(1, f) + "\"");
return r;
}
int main() {
Result r = explainRegex("^(\\w+)@([\\w.-]+)$", "gi");
if (!r.ok) { std::printf("error: %s\n", r.error.c_str()); return 1; }
for (const Token& t : r.tokens) std::printf("%-14s %s\n", t.token.c_str(), t.description.c_str());
for (const auto& [flag, desc] : r.flags) std::printf("flag %c: %s\n", flag, desc.c_str());
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →