Skip to content

Regex Explainer — C++ source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the C++ implementation — the same logic the interactive tool runs, in a shareable, citable form.

// regex-explainer — C++ port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with std::regex
// (ECMAScript grammar — the closest std option to the JS engine; JS /s dotAll has no
// std::regex equivalent, and lookbehind is rejected, so such patterns report ok=false).
#include <cstdio>
#include <map>
#include <regex>
#include <string>
#include <utility>
#include <vector>

struct Token { std::string token, description; };
struct Result {
    bool ok = false;
    std::vector<Token> tokens;
    std::vector<std::pair<char, std::string>> flags;
    std::string error;
};

static const std::map<std::string, std::string> kFlagDesc = {
    {"g", "global - find all matches"}, {"i", "case-insensitive"},
    {"m", "multiline (^ and $ match line boundaries)"}, {"s", "dotAll - \".\" matches newlines"},
    {"u", "unicode"}, {"y", "sticky - match at lastIndex"}, {"d", "indices - expose match boundaries"}};
static const std::map<std::string, std::string> kEscapeDesc = {
    {"d", "a digit [0-9]"}, {"D", "a non-digit"}, {"w", "a word character [A-Za-z0-9_]"},
    {"W", "a non-word character"}, {"s", "a whitespace character"}, {"S", "a non-whitespace character"},
    {"b", "a word boundary"}, {"B", "a non-word boundary"}, {"n", "a newline"},
    {"t", "a tab"}, {"r", "a carriage return"}};

// Index of the ']' closing a class opened at start; a leading ']' is a literal member.
static int findClassEnd(const std::string& p, int start) {
    int i = start + 1;
    if (i < (int)p.size() && p[i] == '^') i++;
    if (i < (int)p.size() && p[i] == ']') i++;
    while (i < (int)p.size() && p[i] != ']') { if (p[i] == '\\') i++; i++; }
    return i < (int)p.size() ? i : (int)p.size() - 1;
}

// Index of the ')' matching the group opened at start; skips classes + escapes.
static int findGroupEnd(const std::string& p, int start) {
    int depth = 1, i = start + 1;
    while (i < (int)p.size() && depth > 0) {
        if (p[i] == '\\') { i += 2; continue; }
        if (p[i] == '[')  { i = findClassEnd(p, i) + 1; continue; }
        if (p[i] == '(') depth++;
        else if (p[i] == ')') depth--;
        i++;
    }
    return i - 1;
}

static std::string describeGroup(const std::string& grp) {
    for (auto& [pre, label] : {std::pair{"(?:", "non-capturing group"},
        {"(?=", "lookahead assertion (positive)"}, {"(?!", "lookahead assertion (negative)"},
        {"(?<=", "lookbehind assertion (positive)"}, {"(?<!", "lookbehind assertion (negative)"}})
        if (grp.rfind(pre, 0) == 0) return label; // portable starts_with
    return "capturing group";
}

static std::string describeClass(const std::string& inner) {
    if (inner.empty()) return "(empty)";
    std::string out;
    for (char c : inner) { if (c == '\\') out += "\\\\"; out += c; } // double '\' for display
    return out;
}

static Result explainRegex(const std::string& pattern, const std::string& flags) {
    auto opts = std::regex::ECMAScript; // JS i/m map onto std bits; /s has no equivalent
    if (flags.find('i') != std::string::npos) opts |= std::regex::icase;
    if (flags.find('m') != std::string::npos) opts |= std::regex::multiline;
    try {
        std::regex re(pattern, opts); // validate with the native engine first
    } catch (const std::regex_error& e) {
        return Result{false, {}, {}, e.what()};
    }

    Result r{true, {}, {}, {}};
    auto push = [&](std::string tok, std::string desc) {
        r.tokens.push_back(Token{std::move(tok), std::move(desc)});
    };
    int i = 0;
    while (i < (int)pattern.size()) {
        char ch = pattern[i];
        switch (ch) {
        case '^': push("^", "start of the string (or line with /m)"); i++; break;
        case '$': push("$", "end of the string (or line with /m)"); i++; break;
        case '.': push(".", "any character (except newline, unless /s)"); i++; break;
        case '|': push("|", "OR - alternation between groups"); i++; break;
        case '\\': {
            std::string nxt = i + 1 < (int)pattern.size() ? std::string(1, pattern[i + 1]) : "";
            auto it = kEscapeDesc.find(nxt);
            push("\\" + nxt, it != kEscapeDesc.end() ? it->second
                                                     : "an escaped literal \"" + nxt + "\"");
            i += 2;
            break;
        }
        case '[': {
            int end = findClassEnd(pattern, i);
            std::string cls = pattern.substr(i, end - i + 1);
            bool negated = pattern[i + 1] == '^';
            std::string inner = cls.substr(1 + (negated ? 1 : 0),
                                           cls.size() - 2 - (negated ? 1 : 0));
            push(cls, "match any " + std::string(negated ? "character NOT in" : "of") + ": " +
                          describeClass(inner));
            i = end + 1;
            break;
        }
        case '(': {
            int end = findGroupEnd(pattern, i);
            std::string grp = pattern.substr(i, end - i + 1);
            push(grp, describeGroup(grp));
            i = end + 1;
            break;
        }
        case '*': case '+': case '?': {
            bool lazy = pattern[i + 1] == '?';
            const char* base = ch == '*' ? "0 or more times"
                             : ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
            push(std::string(1, ch) + (lazy ? "?" : ""),
                 std::string("quantifier - ") + base + (lazy ? " (lazy/non-greedy)" : " (greedy)"));
            i += lazy ? 2 : 1;
            break;
        }
        case '{': {
            size_t close = pattern.find('}', i);
            if (close != std::string::npos) { // bounded quantifier {n,m}
                bool lazy = close + 1 < pattern.size() && pattern[close + 1] == '?';
                std::string q = pattern.substr(i, close - i + 1);
                push(q + (lazy ? "?" : ""), "quantifier - repeat " + q.substr(1, q.size() - 2) +
                                                " time(s)" + (lazy ? " (lazy)" : ""));
                i = (int)close + 1 + (lazy ? 1 : 0);
            } else { // no closing brace: a literal '{'
                push("{", "the literal \"{\"");
                i++;
            }
            break;
        }
        default:
            push(std::string(1, ch), "the literal \"" + std::string(1, ch) + "\"");
            i++;
        }
    }
    for (char f : flags)
        r.flags.emplace_back(f, kFlagDesc.count(std::string(1, f))
                                    ? kFlagDesc.at(std::string(1, f))
                                    : "unknown flag \"" + std::string(1, f) + "\"");
    return r;
}

int main() {
    Result r = explainRegex("^(\\w+)@([\\w.-]+)$", "gi");
    if (!r.ok) { std::printf("error: %s\n", r.error.c_str()); return 1; }
    for (const Token& t : r.tokens) std::printf("%-14s %s\n", t.token.c_str(), t.description.c_str());
    for (const auto& [flag, desc] : r.flags) std::printf("flag %c: %s\n", flag, desc.c_str());
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →