Regex Explainer — Zig source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.
// regex-explainer — Zig port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Zig's std has no regex engine, so
// this is the pure tokenizer half of the tool: any structurally well-formed pattern
// is walked and labeled — validation is left to the embedder's engine of choice
// (the TS lib validates before tokenizing, so its own tokenizer is similarly only
// reached by compilable patterns). Walks bytes, not UTF-16 code units: fine for ASCII.
const std = @import("std");
const Token = struct { token: []const u8, description: []const u8 };
const Flag = struct { flag: u8, description: []const u8 };
const Result = struct { tokens: []Token, flags: []Flag };
fn escapeDesc(c: u8) ?[]const u8 {
return switch (c) {
'd' => "a digit [0-9]", 'D' => "a non-digit", 'w' => "a word character [A-Za-z0-9_]",
'W' => "a non-word character", 's' => "a whitespace character", 'S' => "a non-whitespace character",
'b' => "a word boundary", 'B' => "a non-word boundary", 'n' => "a newline",
't' => "a tab", 'r' => "a carriage return", else => null,
};
}
fn flagDesc(c: u8) ?[]const u8 {
return switch (c) {
'g' => "global - find all matches", 'i' => "case-insensitive",
'm' => "multiline (^ and $ match line boundaries)", 's' => "dotAll - \".\" matches newlines",
'u' => "unicode", 'y' => "sticky - match at lastIndex",
'd' => "indices - expose match boundaries", else => null,
};
}
/// Index of the ']' closing a class opened at start; a leading ']' is a literal member.
fn findClassEnd(p: []const u8, start: usize) usize {
var i = start + 1;
if (i < p.len and p[i] == '^') i += 1;
if (i < p.len and p[i] == ']') i += 1;
while (i < p.len and p[i] != ']') : (i += 1) {
if (p[i] == '\\') i += 1; // skip the escaped member
}
return if (i < p.len) i else p.len - 1;
}
/// Index of the ')' matching the group opened at start; skips classes + escapes.
fn findGroupEnd(p: []const u8, start: usize) usize {
var depth: usize = 1;
var i = start + 1;
while (i < p.len and depth > 0) {
if (p[i] == '\\') { i += 2; continue; }
if (p[i] == '[') { i = findClassEnd(p, i) + 1; continue; }
if (p[i] == '(') depth += 1 else if (p[i] == ')') depth -= 1;
i += 1;
}
return i - 1;
}
fn describeGroup(grp: []const u8) []const u8 {
const P = struct { pre: []const u8, label: []const u8 };
const prefixes = [_]P{
.{ .pre = "(?:", .label = "non-capturing group" },
.{ .pre = "(?=", .label = "lookahead assertion (positive)" },
.{ .pre = "(?!", .label = "lookahead assertion (negative)" },
.{ .pre = "(?<=", .label = "lookbehind assertion (positive)" },
.{ .pre = "(?<!", .label = "lookbehind assertion (negative)" },
};
for (prefixes) |e| if (std.mem.startsWith(u8, grp, e.pre)) return e.label;
return "capturing group";
}
/// Double backslashes so they display as one literal backslash ('(empty)' for no members).
fn describeClass(alloc: std.mem.Allocator, inner: []const u8) ![]const u8 {
if (inner.len == 0) return "(empty)";
var n: usize = 0;
for (inner) |c| if (c == '\\') { n += 1; };
if (n == 0) return inner;
const buf = try alloc.alloc(u8, inner.len + n);
var k: usize = 0;
for (inner) |c| {
if (c == '\\') { buf[k] = '\\'; k += 1; }
buf[k] = c;
k += 1;
}
return buf;
}
/// Tokenize a pattern + flags. Caller-owned slices; never fails on well-formed input.
fn explainRegex(alloc: std.mem.Allocator, pattern: []const u8, flags: []const u8) !Result {
var tokens = std.ArrayList(Token).init(alloc);
var i: usize = 0;
while (i < pattern.len) {
const ch = pattern[i];
switch (ch) {
'^' => try tokens.append(.{ .token = "^", .description = "start of the string (or line with /m)" }),
'$' => try tokens.append(.{ .token = "$", .description = "end of the string (or line with /m)" }),
'.' => try tokens.append(.{ .token = ".", .description = "any character (except newline, unless /s)" }),
'|' => try tokens.append(.{ .token = "|", .description = "OR - alternation between groups" }),
'\\' => {
const nxt: u8 = if (i + 1 < pattern.len) pattern[i + 1] else 0;
const seq = try std.fmt.allocPrint(alloc, "\\{c}", .{nxt});
const desc = escapeDesc(nxt) orelse
try std.fmt.allocPrint(alloc, "an escaped literal \"{c}\"", .{nxt});
try tokens.append(.{ .token = seq, .description = desc });
i += 2;
continue;
},
'[' => {
const end = findClassEnd(pattern, i);
const negated = i + 1 < pattern.len and pattern[i + 1] == '^';
const from = i + 1 + @as(usize, if (negated) 1 else 0);
const inner = if (end > from) pattern[from..end] else ""; // unclosed class
const desc = try std.fmt.allocPrint(alloc, "match any {s}: {s}", .{
if (negated) "character NOT in" else "of",
try describeClass(alloc, inner),
});
try tokens.append(.{ .token = pattern[i .. end + 1], .description = desc });
i = end + 1;
continue;
},
'(' => {
const end = findGroupEnd(pattern, i);
const grp = pattern[i .. end + 1];
try tokens.append(.{ .token = grp, .description = describeGroup(grp) });
i = end + 1;
continue;
},
'*', '+', '?' => {
const lazy = i + 1 < pattern.len and pattern[i + 1] == '?';
const base = switch (ch) {
'*' => "0 or more times", '+' => "1 or more times", else => "0 or 1 time (optional)",
};
const span = i + 1 + @as(usize, if (lazy) 1 else 0);
const desc = try std.fmt.allocPrint(alloc, "quantifier - {s}{s}", .{
base, if (lazy) " (lazy/non-greedy)" else " (greedy)",
});
try tokens.append(.{ .token = pattern[i..span], .description = desc });
i = span;
continue;
},
'{' => {
if (std.mem.indexOfScalarPos(u8, pattern, i, '}')) |end| { // bounded quantifier {n,m}
const lazy = end + 1 < pattern.len and pattern[end + 1] == '?';
const desc = try std.fmt.allocPrint(alloc, "quantifier - repeat {s} time(s){s}", .{
pattern[i + 1 .. end], if (lazy) " (lazy)" else "",
});
const span = end + 1 + @as(usize, if (lazy) 1 else 0);
try tokens.append(.{ .token = pattern[i..span], .description = desc });
i = span;
} else { // no closing brace: a literal '{'
try tokens.append(.{ .token = "{", .description = "the literal \"{\""});
i += 1;
}
continue;
},
else => { // a literal character ('"' displays as \" like the TS escapeHtmlish)
const shown: []const u8 = if (ch == '"') "\\\"" else pattern[i .. i + 1];
const desc = try std.fmt.allocPrint(alloc, "the literal \"{s}\"", .{shown});
try tokens.append(.{ .token = pattern[i .. i + 1], .description = desc });
i += 1;
},
}
}
var flag_list = std.ArrayList(Flag).init(alloc);
for (flags) |f| {
const d = flagDesc(f) orelse try std.fmt.allocPrint(alloc, "unknown flag \"{c}\"", .{f});
try flag_list.append(.{ .flag = f, .description = d });
}
return .{ .tokens = tokens.items, .flags = flag_list.items };
}
pub fn main() !void {
const alloc = std.heap.page_allocator;
const r = try explainRegex(alloc, "^(\\w+)@([\\w.-]+)$", "gi");
// ^(\w+)@([\w.-]+)$ -> anchor, group, literal '@', group, anchor; flags g + i.
std.debug.assert(r.tokens.len == 5);
std.debug.assert(std.mem.eql(u8, r.tokens[0].description, "start of the string (or line with /m)"));
std.debug.assert(std.mem.eql(u8, r.tokens[1].description, "capturing group"));
std.debug.assert(std.mem.eql(u8, r.tokens[2].description, "the literal \"@\""));
std.debug.assert(std.mem.eql(u8, r.tokens[1].token, "(\\w+)"));
std.debug.assert(r.flags.len == 2 and std.mem.eql(u8, r.flags[0].description, "global - find all matches"));
for (r.tokens) |t| std.debug.print("{s} -> {s}\n", .{ t.token, t.description });
std.debug.print("ok\n", .{});
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →