Skip to content

Regex Explainer — C source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.

/* regex-explainer — C port: tokenize a regex into labeled tokens + describe JS flags. */
/* Mirrors src/lib/regexExplain.ts (canonical TS). Validation uses POSIX regcomp
   (REG_EXTENDED): POSIX ERE lacks \w, \d and lookaround, so the demo pattern stays
   POSIX-syntax — the tokenizer itself still labels the JS-only forms. Walks bytes,
   not UTF-16 code units: fine for ASCII patterns. */
#include <regex.h>
#include <stdio.h>
#include <string.h>

#define MAX_TOKENS 64
typedef struct { char token[64], desc[128]; } tok_t;
static char scratch[128]; /* format space for dynamic descriptions */

static const char *FLAG_KEYS[] = { "g", "i", "m", "s", "u", "y", "d" };
static const char *FLAG_DESC[] = { "global - find all matches", "case-insensitive",
    "multiline (^ and $ match line boundaries)", "dotAll - \".\" matches newlines", "unicode",
    "sticky - match at lastIndex", "indices - expose match boundaries" };
static const char *ESC_KEYS[] = { "d", "D", "w", "W", "s", "S", "b", "B", "n", "t", "r" };
static const char *ESC_DESC[] = { "a digit [0-9]", "a non-digit", "a word character [A-Za-z0-9_]",
    "a non-word character", "a whitespace character", "a non-whitespace character", "a word boundary",
    "a non-word boundary", "a newline", "a tab", "a carriage return" };

static const char *lookup(const char **k, const char **d, int n, char c) {
    for (int i = 0; i < n; i++) if (k[i][0] == c && !k[i][1]) return d[i];
    return NULL;
}

static void push(tok_t *toks, int *n, const char *tok, const char *desc) {
    if (*n < MAX_TOKENS) {
        snprintf(toks[*n].token, 64, "%s", tok);
        snprintf(toks[*n].desc, 128, "%s", desc);
        (*n)++;
    }
}

/* Index of the ']' closing a class opened at i; a leading ']' is a literal member. */
static int find_class_end(const char *p, int i) {
    int len = (int)strlen(p);
    i++;
    if (i < len && p[i] == '^') i++;
    if (i < len && p[i] == ']') i++;
    while (i < len && p[i] != ']') { if (p[i] == '\\') i++; i++; }
    return i < len ? i : len - 1;
}

/* Index of the ')' matching the group opened at start; skips classes + escapes. */
static int find_group_end(const char *p, int start) {
    int len = (int)strlen(p), depth = 1, i = start + 1;
    while (i < len && depth > 0) {
        if (p[i] == '\\') { i += 2; continue; }
        if (p[i] == '[')  { i = find_class_end(p, i) + 1; continue; }
        if (p[i] == '(') depth++;
        else if (p[i] == ')') depth--;
        i++;
    }
    return i - 1;
}

static const char *describe_group(const char *g) {
    return !strncmp(g, "(?:", 3)  ? "non-capturing group"
         : !strncmp(g, "(?=", 3)  ? "lookahead assertion (positive)"
         : !strncmp(g, "(?!", 3)  ? "lookahead assertion (negative)"
         : !strncmp(g, "(?<=", 4) ? "lookbehind assertion (positive)"
         : !strncmp(g, "(?<!", 4) ? "lookbehind assertion (negative)"
         : "capturing group";
}

/* Tokenize pattern into toks; returns the token count. */
static int explain(const char *p, tok_t *toks) {
    int n = 0, len = (int)strlen(p), i = 0;
    while (i < len) {
        char ch = p[i];
        if (ch == '^') { push(toks, &n, "^", "start of the string (or line with /m)"); i++; }
        else if (ch == '$') { push(toks, &n, "$", "end of the string (or line with /m)"); i++; }
        else if (ch == '.') { push(toks, &n, ".", "any character (except newline, unless /s)"); i++; }
        else if (ch == '|') { push(toks, &n, "|", "OR - alternation between groups"); i++; }
        else if (ch == '\\') {
            char nxt = p[i + 1], seq[3] = { '\\', nxt, '\0' };
            const char *d = nxt ? lookup(ESC_KEYS, ESC_DESC, 11, nxt) : NULL;
            if (!d) snprintf(scratch, sizeof scratch, "an escaped literal \"%c\"", nxt);
            push(toks, &n, seq, d ? d : scratch);
            i += 2;
        } else if (ch == '[') {
            int end = find_class_end(p, i), negated = p[i + 1] == '^', from = i + 1 + negated, k = 0;
            char cls[64], shown[80];
            snprintf(cls, sizeof cls, "%.*s", end - i + 1, p + i);
            if (end <= from) strcpy(shown, "(empty)");
            else for (int j = from; j < end && k < 78; j++, k++) { /* double '\' for display */
                if (p[j] == '\\') shown[k++] = '\\';
                shown[k] = p[j];
            }
            shown[k] = '\0';
            snprintf(scratch, sizeof scratch, "match any %s: %s",
                     negated ? "character NOT in" : "of", shown);
            push(toks, &n, cls, scratch);
            i = end + 1;
        } else if (ch == '(') {
            int end = find_group_end(p, i);
            char grp[64];
            snprintf(grp, sizeof grp, "%.*s", end - i + 1, p + i);
            push(toks, &n, grp, describe_group(grp));
            i = end + 1;
        } else if (ch == '*' || ch == '+' || ch == '?') {
            int lazy = p[i + 1] == '?';
            char qt[3] = { ch, lazy ? '?' : '\0', '\0' };
            const char *base = ch == '*' ? "0 or more times"
                             : ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
            snprintf(scratch, sizeof scratch, "quantifier - %s%s", base,
                     lazy ? " (lazy/non-greedy)" : " (greedy)");
            push(toks, &n, qt, scratch);
            i += lazy ? 2 : 1;
        } else if (ch == '{' && strchr(p + i, '}')) { /* bounded quantifier {n,m} */
            int end = (int)(strchr(p + i, '}') - p), lazy = p[end + 1] == '?';
            char qt[48];
            snprintf(qt, sizeof qt, "%.*s%s", end - i + 1, p + i, lazy ? "?" : "");
            snprintf(scratch, sizeof scratch, "quantifier - repeat %.*s time(s)%s",
                     end - i - 1, p + i + 1, lazy ? " (lazy)" : "");
            push(toks, &n, qt, scratch);
            i = end + 1 + lazy;
        } else { /* default: a literal character */
            char lit[2] = { ch, '\0' };
            snprintf(scratch, sizeof scratch, "the literal \"%s\"", ch == '"' ? "\\\"" : lit);
            push(toks, &n, lit, scratch);
            i++;
        }
    }
    return n;
}

int main(void) {
    const char *pattern = "^([a-z0-9_.-]+)@([a-z0-9.-]+\\.[a-z]{2,})$", *flags = "gi";
    regex_t re;
    if (regcomp(&re, pattern, REG_EXTENDED | REG_NOSUB)) { /* validate with the native engine */
        fprintf(stderr, "invalid pattern\n");
        return 1;
    }
    regfree(&re);
    tok_t toks[MAX_TOKENS];
    int n = explain(pattern, toks);
    for (int i = 0; i < n; i++) printf("%-16s %s\n", toks[i].token, toks[i].desc);
    for (const char *f = flags; *f; f++) { /* describe each flag letter */
        const char *d = lookup(FLAG_KEYS, FLAG_DESC, 7, *f);
        if (!d) snprintf(scratch, sizeof scratch, "unknown flag \"%c\"", *f), d = scratch;
        printf("flag %c: %s\n", *f, d);
    }
    return 0;
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →