Skip to content

Regex Tester — Zig source

Test regular expressions against any text with all flags (g/i/m/s/u/y). See every match, its index and capture groups, plus the constructed pattern. 100% in-browser.

This is the Zig implementation — the same logic the interactive tool runs, in a shareable, citable form.

// regex-tester — Zig port. Zig ships no regex engine, so this mirrors the
// tool with a compact backtracking matcher: literals, '.', the classes
// \d \w \s, and the quantifiers * + ? on the preceding atom. Capture groups
// are out of scope for a display snippet; matches report text + index,
// capped at 2000 like the TS source. Runnable: zig run zig.zig
const std = @import("std");

const Kind = enum { literal, any, digit, word, space };
const Token = struct { kind: Kind, ch: u8 = 0, min: u32 = 1, max: u32 = 1 };
const Pattern = struct { tokens: [64]Token = undefined, n: usize = 0 };

fn tokenize(pattern: []const u8) ?Pattern {
    var p = Pattern{};
    var i: usize = 0;
    while (i < pattern.len) {
        if (p.n == p.tokens.len) return null; // pattern too long
        var t = Token{ .kind = .literal, .ch = pattern[i] };
        if (pattern[i] == '.') {
            t = .{ .kind = .any };
        } else if (pattern[i] == '\\' and i + 1 < pattern.len) {
            i += 1;
            t = switch (pattern[i]) {
                'd' => .{ .kind = .digit },
                'w' => .{ .kind = .word },
                's' => .{ .kind = .space },
                else => .{ .kind = .literal, .ch = pattern[i] },
            };
        }
        i += 1;
        if (i < pattern.len) switch (pattern[i]) { // quantifier on the atom
            '*' => { t.min = 0; t.max = std.math.maxInt(u32); i += 1; },
            '+' => { t.max = std.math.maxInt(u32); i += 1; },
            '?' => { t.min = 0; t.max = 1; i += 1; },
            else => {},
        }
        p.tokens[p.n] = t;
        p.n += 1;
    }
    return p;
}
fn atomMatch(t: Token, c: u8) bool {
    return switch (t.kind) {
        .literal => c == t.ch,
        .any => c != '\n',
        .digit => std.ascii.isDigit(c),
        .word => std.ascii.isAlphanumeric(c) or c == '_',
        .space => c == ' ' or c == '\t' or c == '\n' or c == '\r',
    };
}
// Greedy match of tokens[ti..] at text[si..]; returns the end offset on
// success. Longest-first with backtracking to the quantifier's min bound.
fn matchFrom(tokens: []const Token, ti: usize, text: []const u8, si: usize) ?usize {
    if (ti == tokens.len) return si;
    const t = tokens[ti];
    var end = si;
    while (end < text.len and end - si < t.max and atomMatch(t, text[end])) end += 1;
    if (end - si < t.min) return null;
    var i = end;
    while (true) {
        if (matchFrom(tokens, ti + 1, text, i)) |e| return e;
        if (i == si + t.min) return null;
        i -= 1;
    }
}

pub fn main() void {
    const text = "mail@x.com nope bob@test.org";
    const pattern = tokenize("\\w+@\\w+\\.\\w+") orelse return;
    const tokens = pattern.tokens[0..pattern.n];
    var start: usize = 0;
    var count: usize = 0;
    while (start <= text.len and count < 2000) {
        if (matchFrom(tokens, 0, text, start)) |e| {
            count += 1;
            std.debug.print("{d}. \"{s}\" @ {d}\n", .{ count, text[start..e], start });
            start = if (e > start) e else start + 1; // skip zero-width loops
        } else start += 1;
    }
    if (count == 0) std.debug.print("(none)\n", .{});
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →