Text Extractor — Rust source
Pull URLs, emails, IPv4/IPv6 addresses, hashes (MD5/SHA-1/SHA-256/SHA-512), and domains out of logs, headers, or any pasted text.
This is the Rust implementation — the same logic the interactive tool runs, in a shareable, citable form.
//! extract — pull URLs, emails, IPv4/IPv6 addresses, hashes, and domains
//! out of arbitrary text (logs, headers, config).
//!
//! Language: Rust (edition 2021)
//! Source: CosmoDev polyglot showcase port of the Extract tool, ported from
//! src/lib/extract.ts (the canonical TypeScript implementation) and
//! held in lock-step with its Go twin cli/extract/extract.go.
//! License: display source — part of CosmoDev's polyglot tool pages.
//!
//! Design goals:
//! - Pure + deterministic; never panics.
//! - Functionally equivalent to the TS/Go reference: same inputs -> same outputs.
//! - The six per-kind patterns are reproduced VERBATIM from the Go twin so the
//! "lock-step contract" between the two implementations is auditable at a
//! glance.
//!
//! Dependency note: unlike Go / Python / PHP / JS, Rust ships no regex engine
//! in std. Text extraction is exactly what regex is for, and every regex-based
//! tool in this showcase (regex-tester, find-replace, ...) therefore uses the
//! `regex` crate — the de-facto-standard, RE2-derived, backtracking-free engine
//! whose semantics match Go's `regexp` directly. Hand-rolling six patterns
//! (word boundaries, greedy alternation, backtracking-adjacent rules like the
//! domain TLD tail) would risk silent divergence from the twin; using `Regex`
//! with the identical pattern strings keeps the ports provably equivalent. This
//! mirrors how slugify/rust.rs documents its own dependency trade-off — we pick
//! the faithful, idiomatic path and say so plainly.
use std::collections::HashSet;
use std::sync::LazyLock;
use regex::Regex;
/// One of the six canonical extraction kinds. Mirrors the TS `ExtractType`
/// union ('url' | 'email' | 'ipv4' | 'ipv6' | 'hash' | 'domain') and Go's
/// `Type`.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
pub enum Kind {
Url,
Email,
Ipv4,
Ipv6,
Hash,
Domain,
}
impl Kind {
/// Canonical spelling used in the TS/Go string-literal contract.
pub fn as_str(self) -> &'static str {
match self {
Kind::Url => "url",
Kind::Email => "email",
Kind::Ipv4 => "ipv4",
Kind::Ipv6 => "ipv6",
Kind::Hash => "hash",
Kind::Domain => "domain",
}
}
}
/// The canonical kinds in display order — Go's `ExtractTypes` / TS's
/// `EXTRACT_TYPES`.
pub const ALL_KINDS: [Kind; 6] = [
Kind::Url, Kind::Email, Kind::Ipv4, Kind::Ipv6, Kind::Hash, Kind::Domain,
];
/// The per-kind patterns, compiled once and reused. These mirror the `RE`
/// record in src/lib/extract.ts and the `re*` vars in the Go twin VERBATIM:
/// same anchors (`\b`), character classes, and counted repetition. The regex
/// crate is RE2, so every construct used (including `\b`, `(?:...)`, and
/// counted alternation) behaves exactly as it does in Go's `regexp`.
///
/// `LazyLock` (std since 1.80) gives us a one-time, thread-safe compile. The
/// `.unwrap()` is sound: these are static pattern literals known to be valid
/// at compile time, so `Regex::new` cannot return `Err` here.
static RE_URL: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"https?://[^\s]+").unwrap());
static RE_EMAIL: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}").unwrap()
});
static RE_IPV4: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b(?:\d{1,3}\.){3}\d{1,3}\b").unwrap());
/// Intentionally permissive — a hex/colon run — post-filtered by `is_ipv6`.
static RE_IPV6: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"[0-9a-fA-F:]+").unwrap());
/// md5 (32) / sha1 (40) / sha256 (64) / sha512 (128). `\b` keeps each length
/// honest, so a 64-char run does not also match as a leading 32-char hash.
static RE_HASH: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(
r"\b[a-fA-F0-9]{32}\b|\b[a-fA-F0-9]{40}\b|\b[a-fA-F0-9]{64}\b|\b[a-fA-F0-9]{128}\b",
)
.unwrap()
});
static RE_DOMAIN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z]{2,})+\b")
.unwrap()
});
/// Validates a single 1–4 hex-digit IPv6 hextet.
static RE_IPV6_GROUP: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^[0-9a-fA-F]{1,4}$").unwrap());
/// The Go twin of the TS `Record<ExtractType, string[]>`. Every field is
/// always present (an empty `Vec`, never `None`): unselected kinds and empty
/// matches are empty, matching the TS lib's "always all six keys" contract.
/// `Default` derives an all-empty result, which is exactly the empty-input
/// shape.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Result {
pub url: Vec<String>,
pub email: Vec<String>,
pub ipv4: Vec<String>,
pub ipv6: Vec<String>,
pub hash: Vec<String>,
pub domain: Vec<String>,
}
/// Deduplicate values, preserving first-occurrence order. The Go twin of the
/// `uniq()` helper in src/lib/extract.ts. A `HashSet` records what we have
/// already emitted; `insert` returns `false` on a repeat, so order is kept
/// without a separate `contains` scan.
fn uniq(values: &[String]) -> Vec<String> {
let mut seen = HashSet::with_capacity(values.len());
let mut out = Vec::with_capacity(values.len());
for v in values {
if seen.insert(v.clone()) {
out.push(v.clone());
}
}
out
}
/// Every non-overlapping match of `re` in `text`. `find_iter` yields matches
/// in left-to-right, non-overlapping order — the RE2 equivalent of Go's
/// `regexp.FindAllString` and JS's `String.prototype.match` with the global
/// flag.
fn all_matches(re: &Regex, text: &str) -> Vec<String> {
re.find_iter(text).map(|m| m.as_str().to_string()).collect()
}
/// Reports whether a hex/colon run is a plausible IPv6 address: it must
/// contain a colon AND either hold a compressed zero-run (`::`) or be exactly
/// eight groups of 1–4 hex digits. Mirrors `isIpv6()` in src/lib/extract.ts.
fn is_ipv6(run: &str) -> bool {
if !run.contains(':') {
return false;
}
if run.contains("::") {
return true;
}
let groups: Vec<&str> = run.split(':').collect();
if groups.len() != 8 {
return false;
}
groups.iter().all(|g| RE_IPV6_GROUP.is_match(g))
}
/// The domain part (after the last `@`) of a matched email. Mirrors
/// `domainOf()` in src/lib/extract.ts.
fn domain_of(email: &str) -> &str {
match email.rfind('@') {
Some(idx) => &email[idx + '@'.len_utf8()..],
None => email,
}
}
/// Extract pulls every occurrence of the given `types` (default: all six) from
/// `input`. It returns a [`Result`] with one field per kind — always all six,
/// populated only for the selected types. Matches are deduped per kind,
/// preserving first-occurrence order. An email also contributes its domain to
/// the `domain` list when both `Email` and `Domain` are selected. It is the Go
/// twin of `extract()` in src/lib/extract.ts and must agree with it on every
/// shared vector.
pub fn extract(input: &str, types: &[Kind]) -> Result {
let selected: &[Kind] = if types.is_empty() { &ALL_KINDS } else { types };
fn want_ref(slice: &[Kind], k: Kind) -> bool {
slice.contains(&k)
}
let mut out = Result::default();
if want_ref(selected, Kind::Url) {
out.url = uniq(&all_matches(&RE_URL, input));
}
if want_ref(selected, Kind::Email) {
out.email = uniq(&all_matches(&RE_EMAIL, input));
}
if want_ref(selected, Kind::Ipv4) {
out.ipv4 = uniq(&all_matches(&RE_IPV4, input));
}
if want_ref(selected, Kind::Ipv6) {
let filtered: Vec<String> = all_matches(&RE_IPV6, input)
.into_iter()
.filter(|r| is_ipv6(r))
.collect();
out.ipv6 = uniq(&filtered);
}
if want_ref(selected, Kind::Hash) {
out.hash = uniq(&all_matches(&RE_HASH, input));
}
if want_ref(selected, Kind::Domain) {
let mut combined = all_matches(&RE_DOMAIN, input);
// Cross-rule: an email also yields its domain in the domain list.
if want_ref(selected, Kind::Email) {
for e in all_matches(&RE_EMAIL, input) {
combined.push(domain_of(&e).to_string());
}
}
out.domain = uniq(&combined);
}
out
}
// ---------- tests (showcase-only; the canonical suite lives in src/lib) ----------
#[cfg(test)]
mod tests {
use super::*;
// Well-known digests of the empty string (real hash values), shared with
// src/lib/extract.test.ts so the showcase uses identical vectors.
const MD5_EMPTY: &str = "d41d8cd98f00b204e9800998ecf8427e"; // 32
const SHA256_EMPTY: &str = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"; // 64
#[test]
fn extracts_and_dedupes_urls() {
let r = extract("a https://x.com b https://y.com c https://x.com", &[]);
assert_eq!(r.url, ["https://x.com", "https://y.com"]);
}
#[test]
fn emails_and_their_domains() {
// Plus-tags and multi-part-TLD domains are matched; when email AND
// domain are both selected, an email also contributes its domain.
let r = extract("reach a.b+tag@mail.example.co.uk please", &[]);
assert_eq!(r.email, ["a.b+tag@mail.example.co.uk"]);
assert_eq!(r.domain, ["mail.example.co.uk"]);
}
#[test]
fn ipv6_keeps_compressed_rejects_times() {
// `::1` is a compressed zero-run; `12:30:45` has no `::` and only 3
// groups, so it is rejected as a clock, not an address.
let r = extract("loopback ::1 and time 12:30:45 now", &[]);
assert_eq!(r.ipv6, ["::1"]);
}
#[test]
fn hashes_by_length() {
let r = extract(format!("m {MD5_EMPTY} s {SHA256_EMPTY}").as_str(), &[]);
assert_eq!(r.hash, [MD5_EMPTY, SHA256_EMPTY]);
}
#[test]
fn type_selection_returns_only_selected() {
let r = extract("https://x.com and a@b.com", &[Kind::Url]);
assert_eq!(r.url, ["https://x.com"]);
assert!(r.email.is_empty()); // email was not selected
}
}
Also available in 13 other languages
Every CosmoDev tool ships its pure logic in TypeScript (web) and Go (CLI), with authored implementations in a dozen-plus languages — the same contract, ported. Compare all languages side by side →