Regex Explainer — Java source
Translate a regular expression into plain English, step by step. Explains anchors, character classes, quantifiers, groups, escapes, alternation, and flags.
This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.
// regex-explainer — Java port: tokenize a regex into labeled tokens + describe JS flags.
// Mirrors src/lib/regexExplain.ts (canonical TS). Validation compiles with
// java.util.regex.Pattern — near-JS syntax; JS-only constructs report ok=false.
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.regex.Pattern;
import java.util.regex.PatternSyntaxException;
final class RegexExplainer {
record RegexToken(String token, String description) {}
record FlagInfo(String flag, String description) {}
record ExplainResult(boolean ok, List<RegexToken> tokens, List<FlagInfo> flags, String error) {}
static final Map<String, String> FLAG_DESC = Map.ofEntries(
Map.entry("g", "global - find all matches"), Map.entry("i", "case-insensitive"),
Map.entry("m", "multiline (^ and $ match line boundaries)"),
Map.entry("s", "dotAll - \".\" matches newlines"), Map.entry("u", "unicode"),
Map.entry("y", "sticky - match at lastIndex"), Map.entry("d", "indices - expose match boundaries"));
static final Map<String, String> ESCAPE_DESC = Map.ofEntries(
Map.entry("d", "a digit [0-9]"), Map.entry("D", "a non-digit"),
Map.entry("w", "a word character [A-Za-z0-9_]"), Map.entry("W", "a non-word character"),
Map.entry("s", "a whitespace character"), Map.entry("S", "a non-whitespace character"),
Map.entry("b", "a word boundary"), Map.entry("B", "a non-word boundary"),
Map.entry("n", "a newline"), Map.entry("t", "a tab"), Map.entry("r", "a carriage return"));
/** Explain a regex pattern + flags into tokens. Never throws. */
static ExplainResult explainRegex(String pattern, String flags) {
// Validate with the native engine first (JS i/m/s map onto Pattern bits).
int opts = 0;
for (char f : flags.toCharArray()) {
if (f == 'i') opts |= Pattern.CASE_INSENSITIVE | Pattern.UNICODE_CASE;
if (f == 'm') opts |= Pattern.MULTILINE;
if (f == 's') opts |= Pattern.DOTALL;
}
try {
Pattern.compile(pattern, opts);
} catch (PatternSyntaxException e) {
return new ExplainResult(false, List.of(), List.of(), e.getMessage());
}
List<RegexToken> tokens = new ArrayList<>();
int i = 0;
while (i < pattern.length()) {
char ch = pattern.charAt(i);
switch (ch) {
case '^' -> { tokens.add(new RegexToken("^", "start of the string (or line with /m)")); i++; }
case '$' -> { tokens.add(new RegexToken("$", "end of the string (or line with /m)")); i++; }
case '.' -> { tokens.add(new RegexToken(".", "any character (except newline, unless /s)")); i++; }
case '|' -> { tokens.add(new RegexToken("|", "OR - alternation between groups")); i++; }
case '\\' -> {
String next = i + 1 < pattern.length() ? String.valueOf(pattern.charAt(i + 1)) : "";
tokens.add(new RegexToken("\\" + next,
ESCAPE_DESC.getOrDefault(next, "an escaped literal \"" + next + "\"")));
i += 2;
}
case '[' -> {
int end = findClassEnd(pattern, i);
String cls = pattern.substring(i, end + 1);
boolean negated = pattern.charAt(i + 1) == '^';
String inner = cls.substring(1 + (negated ? 1 : 0), cls.length() - 1);
tokens.add(new RegexToken(cls, "match any " + (negated ? "character NOT in" : "of")
+ ": " + describeClass(inner)));
i = end + 1;
}
case '(' -> {
int end = findGroupEnd(pattern, i);
String grp = pattern.substring(i, end + 1);
tokens.add(new RegexToken(grp, describeGroup(grp)));
i = end + 1;
}
case '*', '+', '?' -> {
boolean lazy = pattern.charAt(i + 1) == '?';
String base = ch == '*' ? "0 or more times"
: ch == '+' ? "1 or more times" : "0 or 1 time (optional)";
tokens.add(new RegexToken(String.valueOf(ch) + (lazy ? "?" : ""),
"quantifier - " + base + (lazy ? " (lazy/non-greedy)" : " (greedy)")));
i += lazy ? 2 : 1;
}
case '{' -> {
int end = pattern.indexOf('}', i);
if (end != -1) { // bounded quantifier {n,m}
boolean lazy = end + 1 < pattern.length() && pattern.charAt(end + 1) == '?';
String q = pattern.substring(i, end + 1);
tokens.add(new RegexToken(q + (lazy ? "?" : ""),
"quantifier - repeat " + q.substring(1, q.length() - 1)
+ " time(s)" + (lazy ? " (lazy)" : "")));
i = end + 1 + (lazy ? 1 : 0);
} else { // no closing brace: a literal '{'
tokens.add(new RegexToken("{", "the literal \"{\""));
i++;
}
}
default -> { // a literal character
tokens.add(new RegexToken(String.valueOf(ch), "the literal \"" + ch + "\""));
i++;
}
}
}
List<FlagInfo> flagList = new ArrayList<>();
for (char f : flags.toCharArray())
flagList.add(new FlagInfo(String.valueOf(f), FLAG_DESC.getOrDefault(
String.valueOf(f), "unknown flag \"" + f + "\"")));
return new ExplainResult(true, tokens, flagList, null);
}
/** Index of the ']' closing a class opened at start; a leading ']' is a literal member. */
static int findClassEnd(String p, int start) {
int i = start + 1;
if (i < p.length() && p.charAt(i) == '^') i++;
if (i < p.length() && p.charAt(i) == ']') i++;
while (i < p.length() && p.charAt(i) != ']') { if (p.charAt(i) == '\\') i++; i++; }
return i < p.length() ? i : p.length() - 1;
}
/** Index of the ')' matching the group opened at start; skips classes + escapes. */
static int findGroupEnd(String p, int start) {
int depth = 1, i = start + 1;
while (i < p.length() && depth > 0) {
if (p.charAt(i) == '\\') { i += 2; continue; }
if (p.charAt(i) == '[') { i = findClassEnd(p, i) + 1; continue; }
if (p.charAt(i) == '(') depth++;
else if (p.charAt(i) == ')') depth--;
i++;
}
return i - 1;
}
static String describeClass(String inner) {
if (inner.isEmpty()) return "(empty)";
return inner.replace("\\", "\\\\"); // double '\' for display
}
static String describeGroup(String grp) {
if (grp.startsWith("(?:")) return "non-capturing group";
if (grp.startsWith("(?=")) return "lookahead assertion (positive)";
if (grp.startsWith("(?!")) return "lookahead assertion (negative)";
if (grp.startsWith("(?<=")) return "lookbehind assertion (positive)";
if (grp.startsWith("(?<!")) return "lookbehind assertion (negative)";
return "capturing group";
}
public static void main(String[] args) {
ExplainResult r = explainRegex("^(\\w+)@([\\w.-]+)$", "gi");
if (!r.ok()) { System.out.println("error: " + r.error()); return; }
for (RegexToken t : r.tokens())
System.out.printf("%-14s %s%n", t.token(), t.description());
for (FlagInfo f : r.flags())
System.out.println("flag " + f.flag() + ": " + f.description());
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →