Skip to content

Text Extractor — Rust source

Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.

This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.

//! extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
//! out of arbitrary text (logs, headers, config).
//!
//! Language: Rust (edition 2021)
//! Source:   CosmoDev polyglot showcase port of the Extract tool, ported from
//!           src/lib/extract.ts (the canonical TypeScript implementation) and
//!           held in lock-step with its Go twin cli/extract/extract.go.
//! License:  display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//!   - Pure + deterministic; never panics.
//!   - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//!   - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
//!     "lock-step contract" between the two implementations is auditable at a
//!     glance.
//!
//! Dependency note: unlike Go / Python / PHP / JS, Rust ships no regex engine
//! in std. Text extraction is exactly what regex is for, and every regex-based
//! tool in this showcase (regex-tester, find-replace, ...) therefore uses the
//! `regex` crate — the de-facto-standard, RE2-derived, backtracking-free engine
//! whose semantics match Go's `regexp` directly. Hand-rolling six patterns
//! (word boundaries, greedy alternation, backtracking-adjacent rules like the
//! domain TLD tail) would risk silent divergence from the twin; using `Regex`
//! with the identical pattern strings keeps the ports provably equivalent. This
//! mirrors how slugify/rust.rs documents its own dependency trade-off — we pick
//! the faithful, idiomatic path and say so plainly.

use std::collections::HashSet;
use std::sync::LazyLock;

use regex::Regex;

/// One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
/// union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain') and Go's
/// `Type`.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
pub enum Kind {
    Url,
    Email,
    Ipv4,
    Ipv6,
    Hash,
    Domain,
}

impl Kind {
    /// Canonical spelling used in the TS/Go string-literal contract.
    pub fn as_str(self) -> &'static str {
        match self {
            Kind::Url => "url",
            Kind::Email => "email",
            Kind::Ipv4 => "ipv4",
            Kind::Ipv6 => "ipv6",
            Kind::Hash => "hash",
            Kind::Domain => "domain",
        }
    }
}

/// The canonical kinds in display order — Go's `ExtractTypes` / TS's
/// `EXTRACT_TYPES`.
pub const ALL_KINDS: [Kind; 6] = [
    Kind::Url, Kind::Email, Kind::Ipv4, Kind::Ipv6, Kind::Hash, Kind::Domain,
];

/// The per-kind patterns, compiled once and reused. These mirror the `RE`
/// record in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM:
/// same anchors (`\b`), character classes, and counted repetition. The regex
/// crate is RE2, so every construct used (including `\b`, `(?:...)`, and
/// counted alternation) behaves exactly as it does in Go's `regexp`.
///
/// `LazyLock` (std since 1.80) gives us a one-time, thread-safe compile. The
/// `.unwrap()` is sound: these are static pattern literals known to be valid
/// at compile time, so `Regex::new` cannot return `Err` here.
static RE_URL: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r"https?://[^\s]+").unwrap());
static RE_EMAIL: LazyLock<Regex> = LazyLock::new(|| {
    Regex::new(r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}").unwrap()
});
static RE_IPV4: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r"\b(?:\d{1,3}\.){3}\d{1,3}\b").unwrap());
/// Intentionally permissive — a hex/colon run — post-filtered by `is_ipv6`.
static RE_IPV6: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r"[0-9a-fA-F:]+").unwrap());
/// md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length
/// honest, so a 64-char run does not also match as a leading 32-char hash.
static RE_HASH: LazyLock<Regex> = LazyLock::new(|| {
    Regex::new(
        r"\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b",
    )
    .unwrap()
});
static RE_DOMAIN: LazyLock<Regex> = LazyLock::new(|| {
    Regex::new(r"\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b")
        .unwrap()
});
/// Validates a single 1–4 hex-digit IPv6 hextet.
static RE_IPV6_GROUP: LazyLock<Regex> =
    LazyLock::new(|| Regex::new(r"^[0-9a-fA-F]{1,4}$").unwrap());

/// The Go twin of the TS `Record<ExtractType, string[]>`. Every field is
/// always present (an empty `Vec`, never `None`): unselected kinds and empty
/// matches are empty, matching the TS lib's "always all six keys" contract.
/// `Default` derives an all-empty result, which is exactly the empty-input
/// shape.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Result {
    pub url: Vec<String>,
    pub email: Vec<String>,
    pub ipv4: Vec<String>,
    pub ipv6: Vec<String>,
    pub hash: Vec<String>,
    pub domain: Vec<String>,
}

/// Deduplicate values, preserving first-occurrence order. The Go twin of the
/// `uniq()` helper in src/lib/extract.ts. A `HashSet` records what we have
/// already emitted; `insert` returns `false` on a repeat, so order is kept
/// without a separate `contains` scan.
fn uniq(values: &[String]) -> Vec<String> {
    let mut seen = HashSet::with_capacity(values.len());
    let mut out = Vec::with_capacity(values.len());
    for v in values {
        if seen.insert(v.clone()) {
            out.push(v.clone());
        }
    }
    out
}

/// Every non-overlapping match of `re` in `text`. `find_iter` yields matches
/// in left-to-right, non-overlapping order — the RE2 equivalent of Go's
/// `regexp.FindAllString` and JS's `String.prototype.match` with the global
/// flag.
fn all_matches(re: &Regex, text: &str) -> Vec<String> {
    re.find_iter(text).map(|m| m.as_str().to_string()).collect()
}

/// Reports whether a hex/colon run is a plausible IPv6 address: it must
/// contain a colon AND either hold a compressed zero-run (`::`) or be exactly
/// eight groups of 1–4 hex digits. Mirrors `isIpv6()` in src/lib/extract.ts.
fn is_ipv6(run: &str) -> bool {
    if !run.contains(':') {
        return false;
    }
    if run.contains("::") {
        return true;
    }
    let groups: Vec<&str> = run.split(':').collect();
    if groups.len() != 8 {
        return false;
    }
    groups.iter().all(|g| RE_IPV6_GROUP.is_match(g))
}

/// The domain part (after the last `@`) of a matched email. Mirrors
/// `domainOf()` in src/lib/extract.ts.
fn domain_of(email: &str) -> &str {
    match email.rfind('@') {
        Some(idx) => &email[idx + '@'.len_utf8()..],
        None => email,
    }
}

/// Extract pulls every occurrence of the given `types` (default: all six) from
/// `input`. It returns a [`Result`] with one field per kind — always all six,
/// populated only for the selected types. Matches are deduped per kind,
/// preserving first-occurrence order. An email also contributes its domain to
/// the `domain` list when both `Email` and `Domain` are selected. It is the Go
/// twin of `extract()` in src/lib/extract.ts and must agree with it on every
/// shared vector.
pub fn extract(input: &str, types: &[Kind]) -> Result {
    let selected: &[Kind] = if types.is_empty() { &ALL_KINDS } else { types };

    fn want_ref(slice: &[Kind], k: Kind) -> bool {
        slice.contains(&k)
    }

    let mut out = Result::default();

    if want_ref(selected, Kind::Url) {
        out.url = uniq(&all_matches(&RE_URL, input));
    }
    if want_ref(selected, Kind::Email) {
        out.email = uniq(&all_matches(&RE_EMAIL, input));
    }
    if want_ref(selected, Kind::Ipv4) {
        out.ipv4 = uniq(&all_matches(&RE_IPV4, input));
    }
    if want_ref(selected, Kind::Ipv6) {
        let filtered: Vec<String> = all_matches(&RE_IPV6, input)
            .into_iter()
            .filter(|r| is_ipv6(r))
            .collect();
        out.ipv6 = uniq(&filtered);
    }
    if want_ref(selected, Kind::Hash) {
        out.hash = uniq(&all_matches(&RE_HASH, input));
    }
    if want_ref(selected, Kind::Domain) {
        let mut combined = all_matches(&RE_DOMAIN, input);
        // Cross-rule: an email also yields its domain in the domain list.
        if want_ref(selected, Kind::Email) {
            for e in all_matches(&RE_EMAIL, input) {
                combined.push(domain_of(&e).to_string());
            }
        }
        out.domain = uniq(&combined);
    }

    out
}

// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
    use super::*;

    // Well-known digests of the empty string (real hash values), shared with
    // src/lib/extract.test.ts so the showcase uses identical vectors.
    const MD5_EMPTY: &str = "d41d8cd98f00b204e9800998ecf8427e"; // 32
    const SHA256_EMPTY: &str = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64

    #[test]
    fn extracts_and_dedupes_urls() {
        let r = extract("a https://x.com b https://y.com c https://x.com", &[]);
        assert_eq!(r.url, ["https://x.com", "https://y.com"]);
    }

    #[test]
    fn emails_and_their_domains() {
        // Plus-tags and multi-part-TLD domains are matched; when email AND
        // domain are both selected, an email also contributes its domain.
        let r = extract("reach a.b+tag@mail.example.co.uk please", &[]);
        assert_eq!(r.email, ["a.b+tag@mail.example.co.uk"]);
        assert_eq!(r.domain, ["mail.example.co.uk"]);
    }

    #[test]
    fn ipv6_keeps_compressed_rejects_times() {
        // `::1` is a compressed zero-run; `12:30:45` has no `::` and only 3
        // groups, so it is rejected as a clock, not an address.
        let r = extract("loopback ::1 and time 12:30:45 now", &[]);
        assert_eq!(r.ipv6, ["::1"]);
    }

    #[test]
    fn hashes_by_length() {
        let r = extract(format!("m {MD5_EMPTY} s {SHA256_EMPTY}").as_str(), &[]);
        assert_eq!(r.hash, [MD5_EMPTY, SHA256_EMPTY]);
    }

    #[test]
    fn type_selection_returns_only_selected() {
        let r = extract("https://x.com and a@b.com", &[Kind::Url]);
        assert_eq!(r.url, ["https://x.com"]);
        assert!(r.email.is_empty()); // email was not selected
    }
}

Also available in 13 other languages

Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →