Skip to content

Text Extractor — Java source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the Java implementation — the same logic the interactive tool runs, in a shareable, citable form.

import java.util.ArrayList;
import java.util.Arrays;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;

// extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
// out of arbitrary text (logs, headers, config).
//
// Language: Java (17+, standard library only)
// Source:   CosmoDev polyglot showcase port of the Extract tool, ported from
//           src/lib/extract.ts (the canonical TypeScript implementation) and
//           held in lock-step with its Go twin cli/extract/extract.go.
// License:  display source — part of CosmoDev's polyglot tool pages.
//
// Design goals:
//   - Pure + deterministic; never throws.
//   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//   - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
//     "lock-step contract" between the two implementations is auditable at a
//     glance.
//
// Engine note: java.util.regex is a backtracking engine while Go's regexp is
// RE2, but both implement leftmost-first (Perl-order) alternation and none of
// the patterns hinge on a backtracking-only or RE2-only subtlety, so matches
// agree on every input. Java has no raw string literals, so each pattern is
// escaped only as Java string syntax requires (\b -> \\b) and is otherwise
// character-for-character identical to the Go twin.

public final class Extract {

    private Extract() {
    }

    /** One of the six canonical extraction kinds. Mirrors the TS {@code
     * ExtractType} union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' |
     * 'domain') and Go's {@code Type}. */
    public enum Kind {
        URL("url"), EMAIL("email"), IPV4("ipv4"), IPV6("ipv6"), HASH("hash"), DOMAIN("domain");

        private final String label;

        Kind(String label) {
            this.label = label;
        }

        /** Canonical spelling used in the TS/Go string-literal contract. */
        public String label() {
            return label;
        }
    }

    /** The canonical kinds in display order — Go's {@code ExtractTypes} /
     * TS's {@code EXTRACT_TYPES}. */
    public static final List<Kind> ALL_KINDS =
            List.of(Kind.URL, Kind.EMAIL, Kind.IPV4, Kind.IPV6, Kind.HASH, Kind.DOMAIN);

    // The per-kind patterns, compiled once and reused. They mirror the `RE`
    // record in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM:
    // same anchors (\b), character classes, and counted repetition.
    private static final Pattern RE_URL = Pattern.compile("https?://[^\\s]+");
    private static final Pattern RE_EMAIL =
            Pattern.compile("[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\\.[a-zA-Z]{2,}");
    private static final Pattern RE_IPV4 =
            Pattern.compile("\\b(?:\\d{1,3}\\.){3}\\d{1,3}\\b");
    /** Intentionally permissive — a hex/colon run — post-filtered by
     * {@link #isIpv6}. */
    private static final Pattern RE_IPV6 = Pattern.compile("[0-9a-fA-F:]+");
    /** md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). {@code \b} keeps
     * each length honest, so a 64-char run does not also match as a leading
     * 32-char hash. */
    private static final Pattern RE_HASH = Pattern.compile(
            "\\b[a-fA-F0-9]{32}\\b|\\b[a-fA-F0-9]{40}\\b|\\b[a-fA-F0-9]{64}\\b|\\b[a-fA-F0-9]{128}\\b");
    private static final Pattern RE_DOMAIN = Pattern.compile(
            "\\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\\.[a-zA-Z]{2,})+\\b");
    /** Validates a single 1–4 hex-digit IPv6 hextet; {@code matches()} pins
     * both ends, so the ^/$ anchors in the sibling ports are implied. */
    private static final Pattern RE_IPV6_GROUP = Pattern.compile("[0-9a-fA-F]{1,4}");

    /** The Java twin of the TS {@code Record<ExtractType, string[]>}. Every
     * field is always present (an immutable empty list, never null):
     * unselected kinds and empty matches are empty, matching the TS lib's
     * "always all six keys" contract. */
    public record ExtractResult(
            List<String> url,
            List<String> email,
            List<String> ipv4,
            List<String> ipv6,
            List<String> hash,
            List<String> domain) {
        static ExtractResult empty() {
            return new ExtractResult(List.of(), List.of(), List.of(), List.of(), List.of(), List.of());
        }
    }

    /** Deduplicate values, preserving first-occurrence order. The Java twin
     * of the {@code uniq()} helper in src/lib/extract.ts. {@code add} returns
     * false on a repeat, so order is kept without a separate contains
     * scan. */
    private static List<String> uniq(List<String> values) {
        Set<String> seen = new HashSet<>();
        List<String> out = new ArrayList<>(values.size());
        for (String v : values) {
            if (seen.add(v)) out.add(v);
        }
        return out;
    }

    /** Every non-overlapping match of {@code re} in {@code text}, left to
     * right — the java.util.regex equivalent of Go's
     * {@code regexp.FindAllString} and JS's {@code String.prototype.match}
     * with the global flag. */
    private static List<String> allMatches(Pattern re, String text) {
        List<String> out = new ArrayList<>();
        Matcher m = re.matcher(text);
        while (m.find()) out.add(m.group());
        return out;
    }

    /** Reports whether a hex/colon run is a plausible IPv6 address: it must
     * contain a colon AND either hold a compressed zero-run ({@code ::}) or be
     * exactly eight groups of 1–4 hex digits. Mirrors {@code isIpv6()} in
     * src/lib/extract.ts. */
    public static boolean isIpv6(String run) {
        if (!run.contains(":")) return false;
        if (run.contains("::")) return true;
        // limit -1 keeps empty segments at both ends, exactly like split(':')
        // in the TS/Go/Python siblings — Java's default split drops trailing
        // empties, which would wrongly accept "1:2:3:4:5:6:7:8:".
        String[] groups = run.split(":", -1);
        if (groups.length != 8) return false;
        for (String g : groups) {
            if (!RE_IPV6_GROUP.matcher(g).matches()) return false;
        }
        return true;
    }

    /** The domain part (after the last {@code @}) of a matched email.
     * Mirrors {@code domainOf()} in src/lib/extract.ts. */
    public static String domainOf(String email) {
        int idx = email.lastIndexOf('@');
        return idx < 0 ? email : email.substring(idx + 1);
    }

    /** Extract pulls every occurrence of the given {@code types} (default:
     * all six) from {@code input}. It returns an {@link ExtractResult} with
     * one field per kind — always all six, populated only for the selected
     * types. Matches are deduped per kind, preserving first-occurrence order.
     * An email also contributes its domain to the {@code domain} list when
     * both EMAIL and DOMAIN are selected. It is the Java twin of
     * {@code extract()} in src/lib/extract.ts and must agree with it on every
     * shared vector. */
    public static ExtractResult extract(String input, List<Kind> types) {
        String text = input == null ? "" : input;  // TS does `const text = input ?? ''`
        List<Kind> selected = (types == null || types.isEmpty()) ? ALL_KINDS : types;

        List<String> urls = List.of();
        List<String> emails = List.of();
        List<String> ipv4s = List.of();
        List<String> ipv6s = List.of();
        List<String> hashes = List.of();
        List<String> domains = List.of();

        if (selected.contains(Kind.URL)) urls = uniq(allMatches(RE_URL, text));
        if (selected.contains(Kind.EMAIL)) emails = uniq(allMatches(RE_EMAIL, text));
        if (selected.contains(Kind.IPV4)) ipv4s = uniq(allMatches(RE_IPV4, text));
        if (selected.contains(Kind.IPV6)) {
            List<String> filtered = new ArrayList<>();
            for (String r : allMatches(RE_IPV6, text)) {
                if (isIpv6(r)) filtered.add(r);
            }
            ipv6s = uniq(filtered);
        }
        if (selected.contains(Kind.HASH)) hashes = uniq(allMatches(RE_HASH, text));
        if (selected.contains(Kind.DOMAIN)) {
            List<String> combined = new ArrayList<>(allMatches(RE_DOMAIN, text));
            // Cross-rule: an email also yields its domain in the domain list.
            if (selected.contains(Kind.EMAIL)) {
                for (String e : allMatches(RE_EMAIL, text)) combined.add(domainOf(e));
            }
            domains = uniq(combined);
        }

        return new ExtractResult(urls, emails, ipv4s, ipv6s, hashes, domains);
    }

    // ---------- showcase tests (the canonical suite lives in src/lib) ----------
    // Run the class as a program (`java Extract`) for a five-vector smoke
    // check — the Java analogue of the Rust sibling's #[cfg(test)] module.

    /** Well-known digests of the empty string (real hash values), shared with
     * src/lib/extract.test.ts so the showcase uses identical vectors. */
    private static final String MD5_EMPTY = "d41d8cd98f00b204e9800998ecf8427e";  // 32
    private static final String SHA256_EMPTY =
            "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855";  // 64

    private static void checkEq(List<String> actual, List<String> expected, String field) {
        if (!actual.equals(expected)) {
            throw new IllegalStateException(field + ": expected " + expected + ", got " + actual);
        }
    }

    public static void main(String[] args) {
        // Extracts and dedupes URLs, keeping first-occurrence order.
        ExtractResult r = extract("a https://x.com b https://y.com c https://x.com", null);
        checkEq(r.url(), Arrays.asList("https://x.com", "https://y.com"), "url");

        // Plus-tags and multi-part-TLD domains are matched; an email also
        // contributes its domain when email AND domain are both selected.
        r = extract("reach a.b+tag@mail.example.co.uk please", null);
        checkEq(r.email(), List.of("a.b+tag@mail.example.co.uk"), "email");
        checkEq(r.domain(), List.of("mail.example.co.uk"), "domain");

        // `::1` is a compressed zero-run; `12:30:45` has no `::` and only
        // three groups, so it is rejected as a clock, not an address.
        r = extract("loopback ::1 and time 12:30:45 now", null);
        checkEq(r.ipv6(), List.of("::1"), "ipv6");

        // Hashes by length: md5 (32) and sha256 (64).
        r = extract("m " + MD5_EMPTY + " s " + SHA256_EMPTY, null);
        checkEq(r.hash(), List.of(MD5_EMPTY, SHA256_EMPTY), "hash");

        // Type selection: unselected kinds stay empty.
        r = extract("https://x.com and a@b.com", List.of(Kind.URL));
        checkEq(r.url(), List.of("https://x.com"), "url");
        if (!r.email().isEmpty()) throw new IllegalStateException("email was not selected");

        System.out.println("extract: all showcase tests passed");
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →