Skip to content

Regex Explainer — Java source

Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

// regex-explainer — Java port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with
// java.util.regex.Pattern — near-JS syntax; JS-only constructs report ok=false.
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.regex.Pattern;
import java.util.regex.PatternSyntaxException;

final class RegexExplainer {

    record RegexToken(String token, String description) {}
    record FlagInfo(String flag, String description) {}
    record ExplainResult(boolean ok, List<RegexToken> tokens, List<FlagInfo> flags, String error) {}

    static final Map<String, String> FLAG_DESC = Map.ofEntries(
        Map.entry("g", "global - find all matches"), Map.entry("i", "case-insensitive"),
        Map.entry("m", "multiline (^ and $ match line boundaries)"),
        Map.entry("s", "dotAll - \".\" matches newlines"), Map.entry("u", "unicode"),
        Map.entry("y", "sticky - match at lastIndex"), Map.entry("d", "indices - expose match boundaries"));

    static final Map<String, String> ESCAPE_DESC = Map.ofEntries(
        Map.entry("d", "a digit [0-9]"), Map.entry("D", "a non-digit"),
        Map.entry("w", "a word character [A-Za-z0-9_]"), Map.entry("W", "a non-word character"),
        Map.entry("s", "a whitespace character"), Map.entry("S", "a non-whitespace character"),
        Map.entry("b", "a word boundary"), Map.entry("B", "a non-word boundary"),
        Map.entry("n", "a newline"), Map.entry("t", "a tab"), Map.entry("r", "a carriage return"));

    /** Explain a regex pattern + flags into tokens. Never throws. */
    static ExplainResult explainRegex(String pattern, String flags) {
        // Validate with the native engine first (JS i/m/s map onto Pattern bits).
        int opts = 0;
        for (char f : flags.toCharArray()) {
            if (f == 'i') opts |= Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE;
            if (f == 'm') opts |= Pattern.MULTILINE;
            if (f == 's') opts |= Pattern.DOTALL;
        }
        try {
            Pattern.compile(pattern, opts);
        } catch (PatternSyntaxException e) {
            return new ExplainResult(false, List.of(), List.of(), e.getMessage());
        }

        List<RegexToken> tokens = new ArrayList<>();
        int i = 0;
        while (i < pattern.length()) {
            char ch = pattern.charAt(i);
            switch (ch) {
                case '^' -> { tokens.add(new RegexToken("^", "start of the string (or line with /m)")); i++; }
                case '$' -> { tokens.add(new RegexToken("$", "end of the string (or line with /m)")); i++; }
                case '.' -> { tokens.add(new RegexToken(".", "any character (except newline, unless /s)")); i++; }
                case '|' -> { tokens.add(new RegexToken("|", "OR - alternation between groups")); i++; }
                case '\\' -> {
                    String next = i + 1 < pattern.length() ? String.valueOf(pattern.charAt(i + 1)) : "";
                    tokens.add(new RegexToken("\\" + next,
                        ESCAPE_DESC.getOrDefault(next, "an escaped literal \"" + next + "\"")));
                    i += 2;
                }
                case '[' -> {
                    int end = findClassEnd(pattern, i);
                    String cls = pattern.substring(i, end + 1);
                    boolean negated = pattern.charAt(i + 1) == '^';
                    String inner = cls.substring(1 + (negated ? 1 : 0), cls.length() - 1);
                    tokens.add(new RegexToken(cls, "match any " + (negated ? "character NOT in" : "of")
                        + ": " + describeClass(inner)));
                    i = end + 1;
                }
                case '(' -> {
                    int end = findGroupEnd(pattern, i);
                    String grp = pattern.substring(i, end + 1);
                    tokens.add(new RegexToken(grp, describeGroup(grp)));
                    i = end + 1;
                }
                case '*', '+', '?' -> {
                    boolean lazy = pattern.charAt(i + 1) == '?';
                    String base = ch == '*' ? "0 or more times"
                                : ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
                    tokens.add(new RegexToken(String.valueOf(ch) + (lazy ? "?" : ""),
                        "quantifier - " + base + (lazy ? " (lazy/non-greedy)" : " (greedy)")));
                    i += lazy ? 2 : 1;
                }
                case '{' -> {
                    int end = pattern.indexOf('}', i);
                    if (end != -1) { // bounded quantifier {n,m}
                        boolean lazy = end + 1 < pattern.length() && pattern.charAt(end + 1) == '?';
                        String q = pattern.substring(i, end + 1);
                        tokens.add(new RegexToken(q + (lazy ? "?" : ""),
                            "quantifier - repeat " + q.substring(1, q.length() - 1)
                                + " time(s)" + (lazy ? " (lazy)" : "")));
                        i = end + 1 + (lazy ? 1 : 0);
                    } else { // no closing brace: a literal '{'
                        tokens.add(new RegexToken("{", "the literal \"{\""));
                        i++;
                    }
                }
                default -> { // a literal character
                    tokens.add(new RegexToken(String.valueOf(ch), "the literal \"" + ch + "\""));
                    i++;
                }
            }
        }

        List<FlagInfo> flagList = new ArrayList<>();
        for (char f : flags.toCharArray())
            flagList.add(new FlagInfo(String.valueOf(f), FLAG_DESC.getOrDefault(
                String.valueOf(f), "unknown flag \"" + f + "\"")));
        return new ExplainResult(true, tokens, flagList, null);
    }

    /** Index of the ']' closing a class opened at start; a leading ']' is a literal member. */
    static int findClassEnd(String p, int start) {
        int i = start + 1;
        if (i < p.length() && p.charAt(i) == '^') i++;
        if (i < p.length() && p.charAt(i) == ']') i++;
        while (i < p.length() && p.charAt(i) != ']') { if (p.charAt(i) == '\\') i++; i++; }
        return i < p.length() ? i : p.length() - 1;
    }

    /** Index of the ')' matching the group opened at start; skips classes + escapes. */
    static int findGroupEnd(String p, int start) {
        int depth = 1, i = start + 1;
        while (i < p.length() && depth > 0) {
            if (p.charAt(i) == '\\') { i += 2; continue; }
            if (p.charAt(i) == '[') { i = findClassEnd(p, i) + 1; continue; }
            if (p.charAt(i) == '(') depth++;
            else if (p.charAt(i) == ')') depth--;
            i++;
        }
        return i - 1;
    }

    static String describeClass(String inner) {
        if (inner.isEmpty()) return "(empty)";
        return inner.replace("\\", "\\\\"); // double '\' for display
    }

    static String describeGroup(String grp) {
        if (grp.startsWith("(?:")) return "non-capturing group";
        if (grp.startsWith("(?=")) return "lookahead assertion (positive)";
        if (grp.startsWith("(?!")) return "lookahead assertion (negative)";
        if (grp.startsWith("(?<=")) return "lookbehind assertion (positive)";
        if (grp.startsWith("(?<!")) return "lookbehind assertion (negative)";
        return "capturing group";
    }

    public static void main(String[] args) {
        ExplainResult r = explainRegex("^(\\w+)@([\\w.-]+)$", "gi");
        if (!r.ok()) { System.out.println("error: " + r.error()); return; }
        for (RegexToken t : r.tokens())
            System.out.printf("%-14s %s%n", t.token(), t.description());
        for (FlagInfo f : r.flags())
            System.out.println("flag " + f.flag() + ": " + f.description());
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →