Text Extractor — Java source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.
import java.util.ArrayList;
import java.util.Arrays;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
// extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config).
//
// Language: Java (17+, standard library only)
// Source: CosmoDev polyglot showcase port of the Extract tool, ported from
// src/lib/extract.ts (the canonical TypeScript implementation) and
// held in lock-step with its Go twin cli/extract/extract.go.
// License: display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
// - Pure + deterministic; never throws.
// - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
// - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
// "lock-step contract" between the two implementations is auditable at a
// glance.
//
// Engine note: java.util.regex is a backtracking engine while Go's regexp is
// RE2, but both implement leftmost-first (Perl-order) alternation and none of
// the patterns hinge on a backtracking-only or RE2-only subtlety, so matches
// agree on every input. Java has no raw string literals, so each pattern is
// escaped only as Java string syntax requires (\b -> \\b) and is otherwise
// character-for-character identical to the Go twin.
public final class Extract {
private Extract() {
}
/** One of the six canonical extraction kinds. Mirrors the TS {@code
* ExtractType} union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' |
* 'domain') and Go's {@code Type}. */
public enum Kind {
URL("url"), EMAIL("email"), IPV4("ipv4"), IPV6("ipv6"), HASH("hash"), DOMAIN("domain");
private final String label;
Kind(String label) {
this.label = label;
}
/** Canonical spelling used in the TS/Go string-literal contract. */
public String label() {
return label;
}
}
/** The canonical kinds in display order — Go's {@code ExtractTypes} /
* TS's {@code EXTRACT_TYPES}. */
public static final List<Kind> ALL_KINDS =
List.of(Kind.URL, Kind.EMAIL, Kind.IPV4, Kind.IPV6, Kind.HASH, Kind.DOMAIN);
// The per-kind patterns, compiled once and reused. They mirror the `RE`
// record in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM:
// same anchors (\b), character classes, and counted repetition.
private static final Pattern RE_URL = Pattern.compile("https?://[^\\s]+");
private static final Pattern RE_EMAIL =
Pattern.compile("[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\\.[a-zA-Z]{2,}");
private static final Pattern RE_IPV4 =
Pattern.compile("\\b(?:\\d{1,3}\\.){3}\\d{1,3}\\b");
/** Intentionally permissive — a hex/colon run — post-filtered by
* {@link #isIpv6}. */
private static final Pattern RE_IPV6 = Pattern.compile("[0-9a-fA-F:]+");
/** md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). {@code \b} keeps
* each length honest, so a 64-char run does not also match as a leading
* 32-char hash. */
private static final Pattern RE_HASH = Pattern.compile(
"\\b[a-fA-F0-9]{32}\\b|\\b[a-fA-F0-9]{40}\\b|\\b[a-fA-F0-9]{64}\\b|\\b[a-fA-F0-9]{128}\\b");
private static final Pattern RE_DOMAIN = Pattern.compile(
"\\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\\.[a-zA-Z]{2,})+\\b");
/** Validates a single 1–4 hex-digit IPv6 hextet; {@code matches()} pins
* both ends, so the ^/$ anchors in the sibling ports are implied. */
private static final Pattern RE_IPV6_GROUP = Pattern.compile("[0-9a-fA-F]{1,4}");
/** The Java twin of the TS {@code Record<ExtractType, string[]>}. Every
* field is always present (an immutable empty list, never null):
* unselected kinds and empty matches are empty, matching the TS lib's
* "always all six keys" contract. */
public record ExtractResult(
List<String> url,
List<String> email,
List<String> ipv4,
List<String> ipv6,
List<String> hash,
List<String> domain) {
static ExtractResult empty() {
return new ExtractResult(List.of(), List.of(), List.of(), List.of(), List.of(), List.of());
}
}
/** Deduplicate values, preserving first-occurrence order. The Java twin
* of the {@code uniq()} helper in src/lib/extract.ts. {@code add} returns
* false on a repeat, so order is kept without a separate contains
* scan. */
private static List<String> uniq(List<String> values) {
Set<String> seen = new HashSet<>();
List<String> out = new ArrayList<>(values.size());
for (String v : values) {
if (seen.add(v)) out.add(v);
}
return out;
}
/** Every non-overlapping match of {@code re} in {@code text}, left to
* right — the java.util.regex equivalent of Go's
* {@code regexp.FindAllString} and JS's {@code String.prototype.match}
* with the global flag. */
private static List<String> allMatches(Pattern re, String text) {
List<String> out = new ArrayList<>();
Matcher m = re.matcher(text);
while (m.find()) out.add(m.group());
return out;
}
/** Reports whether a hex/colon run is a plausible IPv6 address: it must
* contain a colon AND either hold a compressed zero-run ({@code ::}) or be
* exactly eight groups of 1–4 hex digits. Mirrors {@code isIpv6()} in
* src/lib/extract.ts. */
public static boolean isIpv6(String run) {
if (!run.contains(":")) return false;
if (run.contains("::")) return true;
// limit -1 keeps empty segments at both ends, exactly like split(':')
// in the TS/Go/Python siblings — Java's default split drops trailing
// empties, which would wrongly accept "1:2:3:4:5:6:7:8:".
String[] groups = run.split(":", -1);
if (groups.length != 8) return false;
for (String g : groups) {
if (!RE_IPV6_GROUP.matcher(g).matches()) return false;
}
return true;
}
/** The domain part (after the last {@code @}) of a matched email.
* Mirrors {@code domainOf()} in src/lib/extract.ts. */
public static String domainOf(String email) {
int idx = email.lastIndexOf('@');
return idx < 0 ? email : email.substring(idx + 1);
}
/** Extract pulls every occurrence of the given {@code types} (default:
* all six) from {@code input}. It returns an {@link ExtractResult} with
* one field per kind — always all six, populated only for the selected
* types. Matches are deduped per kind, preserving first-occurrence order.
* An email also contributes its domain to the {@code domain} list when
* both EMAIL and DOMAIN are selected. It is the Java twin of
* {@code extract()} in src/lib/extract.ts and must agree with it on every
* shared vector. */
public static ExtractResult extract(String input, List<Kind> types) {
String text = input == null ? "" : input; // TS does `const text = input ?? ''`
List<Kind> selected = (types == null || types.isEmpty()) ? ALL_KINDS : types;
List<String> urls = List.of();
List<String> emails = List.of();
List<String> ipv4s = List.of();
List<String> ipv6s = List.of();
List<String> hashes = List.of();
List<String> domains = List.of();
if (selected.contains(Kind.URL)) urls = uniq(allMatches(RE_URL, text));
if (selected.contains(Kind.EMAIL)) emails = uniq(allMatches(RE_EMAIL, text));
if (selected.contains(Kind.IPV4)) ipv4s = uniq(allMatches(RE_IPV4, text));
if (selected.contains(Kind.IPV6)) {
List<String> filtered = new ArrayList<>();
for (String r : allMatches(RE_IPV6, text)) {
if (isIpv6(r)) filtered.add(r);
}
ipv6s = uniq(filtered);
}
if (selected.contains(Kind.HASH)) hashes = uniq(allMatches(RE_HASH, text));
if (selected.contains(Kind.DOMAIN)) {
List<String> combined = new ArrayList<>(allMatches(RE_DOMAIN, text));
// Cross-rule: an email also yields its domain in the domain list.
if (selected.contains(Kind.EMAIL)) {
for (String e : allMatches(RE_EMAIL, text)) combined.add(domainOf(e));
}
domains = uniq(combined);
}
return new ExtractResult(urls, emails, ipv4s, ipv6s, hashes, domains);
}
// ---------- showcase tests (the canonical suite lives in src/lib) ----------
// Run the class as a program (`java Extract`) for a five-vector smoke
// check — the Java analogue of the Rust sibling's #[cfg(test)] module.
/** Well-known digests of the empty string (real hash values), shared with
* src/lib/extract.test.ts so the showcase uses identical vectors. */
private static final String MD5_EMPTY = "d41d8cd98f00b204e9800998ecf8427e"; // 32
private static final String SHA256_EMPTY =
"e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64
private static void checkEq(List<String> actual, List<String> expected, String field) {
if (!actual.equals(expected)) {
throw new IllegalStateException(field + ": expected " + expected + ", got " + actual);
}
}
public static void main(String[] args) {
// Extracts and dedupes URLs, keeping first-occurrence order.
ExtractResult r = extract("a https://x.com b https://y.com c https://x.com", null);
checkEq(r.url(), Arrays.asList("https://x.com", "https://y.com"), "url");
// Plus-tags and multi-part-TLD domains are matched; an email also
// contributes its domain when email AND domain are both selected.
r = extract("reach a.b+tag@mail.example.co.uk please", null);
checkEq(r.email(), List.of("a.b+tag@mail.example.co.uk"), "email");
checkEq(r.domain(), List.of("mail.example.co.uk"), "domain");
// `::1` is a compressed zero-run; `12:30:45` has no `::` and only
// three groups, so it is rejected as a clock, not an address.
r = extract("loopback ::1 and time 12:30:45 now", null);
checkEq(r.ipv6(), List.of("::1"), "ipv6");
// Hashes by length: md5 (32) and sha256 (64).
r = extract("m " + MD5_EMPTY + " s " + SHA256_EMPTY, null);
checkEq(r.hash(), List.of(MD5_EMPTY, SHA256_EMPTY), "hash");
// Type selection: unselected kinds stay empty.
r = extract("https://x.com and a@b.com", List.of(Kind.URL));
checkEq(r.url(), List.of("https://x.com"), "url");
if (!r.email().isEmpty()) throw new IllegalStateException("email was not selected");
System.out.println("extract: all showcase tests passed");
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →