Skip to content

Regex Explainer — Zig source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// regex-explainer — Zig port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Zig's std has no regex engine, so
// this is the pure tokenizer half of the tool: any structurally well-formed pattern
// is walked and labeled — validation is left to the embedder's engine of choice
// (the TS lib validates before tokenizing, so its own tokenizer is similarly only
// reached by compilable patterns). Walks bytes, not UTF-16 code units: fine for ASCII.
const std = @import("std");

const Token = struct { token: []const u8, description: []const u8 };
const Flag = struct { flag: u8, description: []const u8 };
const Result = struct { tokens: []Token, flags: []Flag };

fn escapeDesc(c: u8) ?[]const u8 {
    return switch (c) {
        'd' => "a digit [0-9]", 'D' => "a non-digit", 'w' => "a word character [A-Za-z0-9_]",
        'W' => "a non-word character", 's' => "a whitespace character", 'S' => "a non-whitespace character",
        'b' => "a word boundary", 'B' => "a non-word boundary", 'n' => "a newline",
        't' => "a tab", 'r' => "a carriage return", else => null,
    };
}

fn flagDesc(c: u8) ?[]const u8 {
    return switch (c) {
        'g' => "global - find all matches", 'i' => "case-insensitive",
        'm' => "multiline (^ and $ match line boundaries)", 's' => "dotAll - \".\" matches newlines",
        'u' => "unicode", 'y' => "sticky - match at lastIndex",
        'd' => "indices - expose match boundaries", else => null,
    };
}

/// Index of the ']' closing a class opened at start; a leading ']' is a literal member.
fn findClassEnd(p: []const u8, start: usize) usize {
    var i = start + 1;
    if (i < p.len and p[i] == '^') i += 1;
    if (i < p.len and p[i] == ']') i += 1;
    while (i < p.len and p[i] != ']') : (i += 1) {
        if (p[i] == '\\') i += 1; // skip the escaped member
    }
    return if (i < p.len) i else p.len - 1;
}

/// Index of the ')' matching the group opened at start; skips classes + escapes.
fn findGroupEnd(p: []const u8, start: usize) usize {
    var depth: usize = 1;
    var i = start + 1;
    while (i < p.len and depth > 0) {
        if (p[i] == '\\') { i += 2; continue; }
        if (p[i] == '[') { i = findClassEnd(p, i) + 1; continue; }
        if (p[i] == '(') depth += 1 else if (p[i] == ')') depth -= 1;
        i += 1;
    }
    return i - 1;
}

fn describeGroup(grp: []const u8) []const u8 {
    const P = struct { pre: []const u8, label: []const u8 };
    const prefixes = [_]P{
        .{ .pre = "(?:", .label = "non-capturing group" },
        .{ .pre = "(?=", .label = "lookahead assertion (positive)" },
        .{ .pre = "(?!", .label = "lookahead assertion (negative)" },
        .{ .pre = "(?<=", .label = "lookbehind assertion (positive)" },
        .{ .pre = "(?<!", .label = "lookbehind assertion (negative)" },
    };
    for (prefixes) |e| if (std.mem.startsWith(u8, grp, e.pre)) return e.label;
    return "capturing group";
}

/// Double backslashes so they display as one literal backslash ('(empty)' for no members).
fn describeClass(alloc: std.mem.Allocator, inner: []const u8) ![]const u8 {
    if (inner.len == 0) return "(empty)";
    var n: usize = 0;
    for (inner) |c| if (c == '\\') { n += 1; };
    if (n == 0) return inner;
    const buf = try alloc.alloc(u8, inner.len + n);
    var k: usize = 0;
    for (inner) |c| {
        if (c == '\\') { buf[k] = '\\'; k += 1; }
        buf[k] = c;
        k += 1;
    }
    return buf;
}

/// Tokenize a pattern + flags. Caller-owned slices; never fails on well-formed input.
fn explainRegex(alloc: std.mem.Allocator, pattern: []const u8, flags: []const u8) !Result {
    var tokens = std.ArrayList(Token).init(alloc);
    var i: usize = 0;
    while (i < pattern.len) {
        const ch = pattern[i];
        switch (ch) {
            '^' => try tokens.append(.{ .token = "^", .description = "start of the string (or line with /m)" }),
            '$' => try tokens.append(.{ .token = "$", .description = "end of the string (or line with /m)" }),
            '.' => try tokens.append(.{ .token = ".", .description = "any character (except newline, unless /s)" }),
            '|' => try tokens.append(.{ .token = "|", .description = "OR - alternation between groups" }),
            '\\' => {
                const nxt: u8 = if (i + 1 < pattern.len) pattern[i + 1] else 0;
                const seq = try std.fmt.allocPrint(alloc, "\\{c}", .{nxt});
                const desc = escapeDesc(nxt) orelse
                    try std.fmt.allocPrint(alloc, "an escaped literal \"{c}\"", .{nxt});
                try tokens.append(.{ .token = seq, .description = desc });
                i += 2;
                continue;
            },
            '[' => {
                const end = findClassEnd(pattern, i);
                const negated = i + 1 < pattern.len and pattern[i + 1] == '^';
                const from = i + 1 + @as(usize, if (negated) 1 else 0);
                const inner = if (end > from) pattern[from..end] else ""; // unclosed class
                const desc = try std.fmt.allocPrint(alloc, "match any {s}: {s}", .{
                    if (negated) "character NOT in" else "of",
                    try describeClass(alloc, inner),
                });
                try tokens.append(.{ .token = pattern[i .. end + 1], .description = desc });
                i = end + 1;
                continue;
            },
            '(' => {
                const end = findGroupEnd(pattern, i);
                const grp = pattern[i .. end + 1];
                try tokens.append(.{ .token = grp, .description = describeGroup(grp) });
                i = end + 1;
                continue;
            },
            '*', '+', '?' => {
                const lazy = i + 1 < pattern.len and pattern[i + 1] == '?';
                const base = switch (ch) {
                    '*' => "0 or more times", '+' => "1 or more times", else => "0 or 1 time (optional)",
                };
                const span = i + 1 + @as(usize, if (lazy) 1 else 0);
                const desc = try std.fmt.allocPrint(alloc, "quantifier - {s}{s}", .{
                    base, if (lazy) " (lazy/non-greedy)" else " (greedy)",
                });
                try tokens.append(.{ .token = pattern[i..span], .description = desc });
                i = span;
                continue;
            },
            '{' => {
                if (std.mem.indexOfScalarPos(u8, pattern, i, '}')) |end| { // bounded quantifier {n,m}
                    const lazy = end + 1 < pattern.len and pattern[end + 1] == '?';
                    const desc = try std.fmt.allocPrint(alloc, "quantifier - repeat {s} time(s){s}", .{
                        pattern[i + 1 .. end], if (lazy) " (lazy)" else "",
                    });
                    const span = end + 1 + @as(usize, if (lazy) 1 else 0);
                    try tokens.append(.{ .token = pattern[i..span], .description = desc });
                    i = span;
                } else { // no closing brace: a literal '{'
                    try tokens.append(.{ .token = "{", .description = "the literal \"{\""});
                    i += 1;
                }
                continue;
            },
            else => { // a literal character ('"' displays as \" like the TS escapeHtmlish)
                const shown: []const u8 = if (ch == '"') "\\\"" else pattern[i .. i + 1];
                const desc = try std.fmt.allocPrint(alloc, "the literal \"{s}\"", .{shown});
                try tokens.append(.{ .token = pattern[i .. i + 1], .description = desc });
                i += 1;
            },
        }
    }

    var flag_list = std.ArrayList(Flag).init(alloc);
    for (flags) |f| {
        const d = flagDesc(f) orelse try std.fmt.allocPrint(alloc, "unknown flag \"{c}\"", .{f});
        try flag_list.append(.{ .flag = f, .description = d });
    }
    return .{ .tokens = tokens.items, .flags = flag_list.items };
}

pub fn main() !void {
    const alloc = std.heap.page_allocator;
    const r = try explainRegex(alloc, "^(\\w+)@([\\w.-]+)$", "gi");
    // ^(\w+)@([\w.-]+)$ -> anchor, group, literal '@', group, anchor; flags g + i.
    std.debug.assert(r.tokens.len == 5);
    std.debug.assert(std.mem.eql(u8, r.tokens[0].description, "start of the string (or line with /m)"));
    std.debug.assert(std.mem.eql(u8, r.tokens[1].description, "capturing group"));
    std.debug.assert(std.mem.eql(u8, r.tokens[2].description, "the literal \"@\""));
    std.debug.assert(std.mem.eql(u8, r.tokens[1].token, "(\\w+)"));
    std.debug.assert(r.flags.len == 2 and std.mem.eql(u8, r.flags[0].description, "global - find all matches"));
    for (r.tokens) |t| std.debug.print("{s} -> {s}\n", .{ t.token, t.description });
    std.debug.print("ok\n", .{});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →