Regex Explainer — C source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the C implementation — the same logic the interactive tool runs, in a shareable, citable form.
/* regex-explainer — C port: tokenize a regex into labeled tokens + describe JS flags. */
/* Mirrors src/lib/regexExplain.ts (canonical TS). Validation uses POSIX regcomp
(REG_EXTENDED): POSIX ERE lacks \w, \d and lookaround, so the demo pattern stays
POSIX-syntax — the tokenizer itself still labels the JS-only forms. Walks bytes,
not UTF-16 code units: fine for ASCII patterns. */
#include <regex.h>
#include <stdio.h>
#include <string.h>
#define MAX_TOKENS 64
typedef struct { char token[64], desc[128]; } tok_t;
static char scratch[128]; /* format space for dynamic descriptions */
static const char *FLAG_KEYS[] = { "g", "i", "m", "s", "u", "y", "d" };
static const char *FLAG_DESC[] = { "global - find all matches", "case-insensitive",
"multiline (^ and $ match line boundaries)", "dotAll - \".\" matches newlines", "unicode",
"sticky - match at lastIndex", "indices - expose match boundaries" };
static const char *ESC_KEYS[] = { "d", "D", "w", "W", "s", "S", "b", "B", "n", "t", "r" };
static const char *ESC_DESC[] = { "a digit [0-9]", "a non-digit", "a word character [A-Za-z0-9_]",
"a non-word character", "a whitespace character", "a non-whitespace character", "a word boundary",
"a non-word boundary", "a newline", "a tab", "a carriage return" };
static const char *lookup(const char **k, const char **d, int n, char c) {
for (int i = 0; i < n; i++) if (k[i][0] == c && !k[i][1]) return d[i];
return NULL;
}
static void push(tok_t *toks, int *n, const char *tok, const char *desc) {
if (*n < MAX_TOKENS) {
snprintf(toks[*n].token, 64, "%s", tok);
snprintf(toks[*n].desc, 128, "%s", desc);
(*n)++;
}
}
/* Index of the ']' closing a class opened at i; a leading ']' is a literal member. */
static int find_class_end(const char *p, int i) {
int len = (int)strlen(p);
i++;
if (i < len && p[i] == '^') i++;
if (i < len && p[i] == ']') i++;
while (i < len && p[i] != ']') { if (p[i] == '\\') i++; i++; }
return i < len ? i : len - 1;
}
/* Index of the ')' matching the group opened at start; skips classes + escapes. */
static int find_group_end(const char *p, int start) {
int len = (int)strlen(p), depth = 1, i = start + 1;
while (i < len && depth > 0) {
if (p[i] == '\\') { i += 2; continue; }
if (p[i] == '[') { i = find_class_end(p, i) + 1; continue; }
if (p[i] == '(') depth++;
else if (p[i] == ')') depth--;
i++;
}
return i - 1;
}
static const char *describe_group(const char *g) {
return !strncmp(g, "(?:", 3) ? "non-capturing group"
: !strncmp(g, "(?=", 3) ? "lookahead assertion (positive)"
: !strncmp(g, "(?!", 3) ? "lookahead assertion (negative)"
: !strncmp(g, "(?<=", 4) ? "lookbehind assertion (positive)"
: !strncmp(g, "(?<!", 4) ? "lookbehind assertion (negative)"
: "capturing group";
}
/* Tokenize pattern into toks; returns the token count. */
static int explain(const char *p, tok_t *toks) {
int n = 0, len = (int)strlen(p), i = 0;
while (i < len) {
char ch = p[i];
if (ch == '^') { push(toks, &n, "^", "start of the string (or line with /m)"); i++; }
else if (ch == '$') { push(toks, &n, "$", "end of the string (or line with /m)"); i++; }
else if (ch == '.') { push(toks, &n, ".", "any character (except newline, unless /s)"); i++; }
else if (ch == '|') { push(toks, &n, "|", "OR - alternation between groups"); i++; }
else if (ch == '\\') {
char nxt = p[i + 1], seq[3] = { '\\', nxt, '\0' };
const char *d = nxt ? lookup(ESC_KEYS, ESC_DESC, 11, nxt) : NULL;
if (!d) snprintf(scratch, sizeof scratch, "an escaped literal \"%c\"", nxt);
push(toks, &n, seq, d ? d : scratch);
i += 2;
} else if (ch == '[') {
int end = find_class_end(p, i), negated = p[i + 1] == '^', from = i + 1 + negated, k = 0;
char cls[64], shown[80];
snprintf(cls, sizeof cls, "%.*s", end - i + 1, p + i);
if (end <= from) strcpy(shown, "(empty)");
else for (int j = from; j < end && k < 78; j++, k++) { /* double '\' for display */
if (p[j] == '\\') shown[k++] = '\\';
shown[k] = p[j];
}
shown[k] = '\0';
snprintf(scratch, sizeof scratch, "match any %s: %s",
negated ? "character NOT in" : "of", shown);
push(toks, &n, cls, scratch);
i = end + 1;
} else if (ch == '(') {
int end = find_group_end(p, i);
char grp[64];
snprintf(grp, sizeof grp, "%.*s", end - i + 1, p + i);
push(toks, &n, grp, describe_group(grp));
i = end + 1;
} else if (ch == '*' || ch == '+' || ch == '?') {
int lazy = p[i + 1] == '?';
char qt[3] = { ch, lazy ? '?' : '\0', '\0' };
const char *base = ch == '*' ? "0 or more times"
: ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
snprintf(scratch, sizeof scratch, "quantifier - %s%s", base,
lazy ? " (lazy/non-greedy)" : " (greedy)");
push(toks, &n, qt, scratch);
i += lazy ? 2 : 1;
} else if (ch == '{' && strchr(p + i, '}')) { /* bounded quantifier {n,m} */
int end = (int)(strchr(p + i, '}') - p), lazy = p[end + 1] == '?';
char qt[48];
snprintf(qt, sizeof qt, "%.*s%s", end - i + 1, p + i, lazy ? "?" : "");
snprintf(scratch, sizeof scratch, "quantifier - repeat %.*s time(s)%s",
end - i - 1, p + i + 1, lazy ? " (lazy)" : "");
push(toks, &n, qt, scratch);
i = end + 1 + lazy;
} else { /* default: a literal character */
char lit[2] = { ch, '\0' };
snprintf(scratch, sizeof scratch, "the literal \"%s\"", ch == '"' ? "\\\"" : lit);
push(toks, &n, lit, scratch);
i++;
}
}
return n;
}
int main(void) {
const char *pattern = "^([a-z0-9_.-]+)@([a-z0-9.-]+\\.[a-z]{2,})$", *flags = "gi";
regex_t re;
if (regcomp(&re, pattern, REG_EXTENDED | REG_NOSUB)) { /* validate with the native engine */
fprintf(stderr, "invalid pattern\n");
return 1;
}
regfree(&re);
tok_t toks[MAX_TOKENS];
int n = explain(pattern, toks);
for (int i = 0; i < n; i++) printf("%-16s %s\n", toks[i].token, toks[i].desc);
for (const char *f = flags; *f; f++) { /* describe each flag letter */
const char *d = lookup(FLAG_KEYS, FLAG_DESC, 7, *f);
if (!d) snprintf(scratch, sizeof scratch, "unknown flag \"%c\"", *f), d = scratch;
printf("flag %c: %s\n", *f, d);
}
return 0;
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →